@anvia/core 1.0.0-rc.1 → 1.0.0-rc.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +344 -85
- package/dist/agent/index.d.ts +33 -18
- package/dist/agent/index.js +42 -17
- package/dist/agent-CXScUOpX.d.ts +150 -0
- package/dist/{chunk-R2LJSQUW.js → chunk-3CMWKK32.js} +260 -104
- package/dist/chunk-3CMWKK32.js.map +1 -0
- package/dist/chunk-3RM57ZT2.js +31 -0
- package/dist/chunk-3RM57ZT2.js.map +1 -0
- package/dist/chunk-5MK3BISB.js +704 -0
- package/dist/chunk-5MK3BISB.js.map +1 -0
- package/dist/chunk-AI5JMUW7.js +45 -0
- package/dist/chunk-AI5JMUW7.js.map +1 -0
- package/dist/{chunk-FNQB2JEH.js → chunk-AZB6N7P4.js} +94 -65
- package/dist/chunk-AZB6N7P4.js.map +1 -0
- package/dist/chunk-BEYBU7VJ.js +53 -0
- package/dist/chunk-BEYBU7VJ.js.map +1 -0
- package/dist/chunk-BY4OMNIU.js +4699 -0
- package/dist/chunk-BY4OMNIU.js.map +1 -0
- package/dist/chunk-D2ECOWN5.js +515 -0
- package/dist/chunk-D2ECOWN5.js.map +1 -0
- package/dist/{chunk-A3UBYTJT.js → chunk-FN7HLLAZ.js} +60 -6
- package/dist/chunk-FN7HLLAZ.js.map +1 -0
- package/dist/chunk-GKDERULJ.js +50 -0
- package/dist/chunk-GKDERULJ.js.map +1 -0
- package/dist/chunk-GQL5KRAH.js +331 -0
- package/dist/chunk-GQL5KRAH.js.map +1 -0
- package/dist/{chunk-5XKDW36Z.js → chunk-KFR4CDGK.js} +70 -75
- package/dist/chunk-KFR4CDGK.js.map +1 -0
- package/dist/chunk-KV6QCRQ4.js +171 -0
- package/dist/chunk-KV6QCRQ4.js.map +1 -0
- package/dist/{chunk-DMPUP4R3.js → chunk-LXMAFD23.js} +5 -5
- package/dist/{chunk-I6G7BC42.js → chunk-MGNTXGTX.js} +2 -6
- package/dist/chunk-MGNTXGTX.js.map +1 -0
- package/dist/chunk-PK5LKOFL.js +239 -0
- package/dist/chunk-PK5LKOFL.js.map +1 -0
- package/dist/chunk-QOUPWOGW.js +29 -0
- package/dist/chunk-QOUPWOGW.js.map +1 -0
- package/dist/chunk-TRQ3XRFD.js +54 -0
- package/dist/chunk-TRQ3XRFD.js.map +1 -0
- package/dist/chunk-VKYZZXP5.js +323 -0
- package/dist/chunk-VKYZZXP5.js.map +1 -0
- package/dist/chunk-ZCPOIDAQ.js +127 -0
- package/dist/chunk-ZCPOIDAQ.js.map +1 -0
- package/dist/client-DvUatElp.d.ts +84 -0
- package/dist/completion/index.d.ts +5 -4
- package/dist/completion/index.js +25 -28
- package/dist/documents/index.d.ts +34 -0
- package/dist/documents/index.js +315 -0
- package/dist/documents/index.js.map +1 -0
- package/dist/{dynamic-tools-i9woysFa.d.ts → dynamic-tools-AEBOTgu1.d.ts} +21 -9
- package/dist/embeddings/index.d.ts +11 -9
- package/dist/embeddings/index.js +3 -3
- package/dist/evals/index.d.ts +53 -25
- package/dist/evals/index.js +153 -103
- package/dist/evals/index.js.map +1 -1
- package/dist/extractor/index.d.ts +23 -26
- package/dist/extractor/index.js +7 -8
- package/dist/guardrails/index.d.ts +7 -129
- package/dist/guardrails/index.js +1 -1
- package/dist/image-generation/index.d.ts +18 -16
- package/dist/image-generation/index.js +2 -2
- package/dist/index.d.ts +26 -23
- package/dist/index.js +71 -48
- package/dist/interactions-csssHYcN.d.ts +174 -0
- package/dist/internal/agent.d.ts +36 -29
- package/dist/internal/agent.js +48 -15
- package/dist/internal/agent.js.map +1 -1
- package/dist/mcp/index.d.ts +19 -13
- package/dist/mcp/index.js +14 -348
- package/dist/mcp/index.js.map +1 -1
- package/dist/memory/index.d.ts +20 -6
- package/dist/memory/index.js +10 -10
- package/dist/message-schema-B2M7AYzR.d.ts +52 -0
- package/dist/model-call-options-CZkSw_xN.d.ts +6 -0
- package/dist/model-listing/index.d.ts +3 -1
- package/dist/observability/index.d.ts +21 -6
- package/dist/observability/index.js +8 -5
- package/dist/observability/index.js.map +1 -1
- package/dist/pipeline/index.d.ts +106 -57
- package/dist/pipeline/index.js +419 -236
- package/dist/pipeline/index.js.map +1 -1
- package/dist/{retry-D3Ruy-ba.d.ts → retry-CjvSlKGW.d.ts} +2 -1
- package/dist/skills/index.d.ts +5 -4
- package/dist/skills/index.js +8 -7
- package/dist/speech-generation/index.d.ts +37 -0
- package/dist/speech-generation/index.js +8 -0
- package/dist/text-C_6eKrbC.d.ts +32 -0
- package/dist/{think-tool-DZqZj2LV.d.ts → think-tool-D8frXxwV.d.ts} +16 -2
- package/dist/tool/index.d.ts +9 -7
- package/dist/tool/index.js +12 -10
- package/dist/{tool-C_8GudXn.d.ts → tool-r62gf3QY.d.ts} +17 -8
- package/dist/transcription/index.d.ts +23 -14
- package/dist/transcription/index.js +2 -2
- package/dist/types-C3gv9EX6.d.ts +130 -0
- package/dist/{index-DxB74ybJ.d.ts → types-CElwurHZ.d.ts} +139 -88
- package/dist/types-CIPaIbZo.d.ts +105 -0
- package/dist/types-Cr4uiYo5.d.ts +99 -0
- package/dist/types-D9rKQ_as.d.ts +341 -0
- package/dist/types-DbolvLtZ.d.ts +155 -0
- package/dist/{types-DhQfwiZn.d.ts → types-RT3cuyRk.d.ts} +1 -1
- package/dist/vector-store/index.d.ts +33 -34
- package/dist/vector-store/index.js +7 -6
- package/package.json +10 -14
- package/dist/agent-DGYY_onS.d.ts +0 -145
- package/dist/audio-generation/index.d.ts +0 -31
- package/dist/audio-generation/index.js +0 -8
- package/dist/chunk-5XKDW36Z.js.map +0 -1
- package/dist/chunk-6WGGTJE6.js +0 -139
- package/dist/chunk-6WGGTJE6.js.map +0 -1
- package/dist/chunk-A3UBYTJT.js.map +0 -1
- package/dist/chunk-ADH7NNCS.js +0 -512
- package/dist/chunk-ADH7NNCS.js.map +0 -1
- package/dist/chunk-DZ3CKYTH.js +0 -267
- package/dist/chunk-DZ3CKYTH.js.map +0 -1
- package/dist/chunk-FNQB2JEH.js.map +0 -1
- package/dist/chunk-I6G7BC42.js.map +0 -1
- package/dist/chunk-IL3ZSNJI.js +0 -3748
- package/dist/chunk-IL3ZSNJI.js.map +0 -1
- package/dist/chunk-J6GNQART.js +0 -122
- package/dist/chunk-J6GNQART.js.map +0 -1
- package/dist/chunk-M3IOWB4Y.js +0 -280
- package/dist/chunk-M3IOWB4Y.js.map +0 -1
- package/dist/chunk-QY5GZ7HR.js +0 -39
- package/dist/chunk-QY5GZ7HR.js.map +0 -1
- package/dist/chunk-R2LJSQUW.js.map +0 -1
- package/dist/chunk-RFDVKCIE.js +0 -12
- package/dist/chunk-RFDVKCIE.js.map +0 -1
- package/dist/chunk-RMUBRRSK.js +0 -34
- package/dist/chunk-RMUBRRSK.js.map +0 -1
- package/dist/chunk-RXKHOYEH.js +0 -466
- package/dist/chunk-RXKHOYEH.js.map +0 -1
- package/dist/chunk-XDRSPSR7.js +0 -38
- package/dist/chunk-XDRSPSR7.js.map +0 -1
- package/dist/chunk-XWUC7CIT.js +0 -1
- package/dist/chunk-XWUC7CIT.js.map +0 -1
- package/dist/errors-CYKUcU89.d.ts +0 -17
- package/dist/json-Zo7K3xmt.d.ts +0 -40
- package/dist/loaders/index.d.ts +0 -86
- package/dist/loaders/index.js +0 -299
- package/dist/loaders/index.js.map +0 -1
- package/dist/middleware-KynLg_RW.d.ts +0 -56
- package/dist/types-B3A4YxaA.d.ts +0 -354
- package/dist/types-BBEGKB9m.d.ts +0 -119
- package/dist/types-DT3nEemY.d.ts +0 -79
- package/dist/types-DulRCYdB.d.ts +0 -70
- package/dist/types-p55Jc-sr.d.ts +0 -47
- package/dist/ui/index.d.ts +0 -107
- package/dist/ui/index.js +0 -10
- package/dist/ui/index.js.map +0 -1
- /package/dist/{chunk-DMPUP4R3.js.map → chunk-LXMAFD23.js.map} +0 -0
- /package/dist/{audio-generation → speech-generation}/index.js.map +0 -0
package/dist/evals/index.d.ts
CHANGED
|
@@ -1,22 +1,15 @@
|
|
|
1
|
-
import { U as Usage,
|
|
1
|
+
import { U as Usage, J as JsonObject, C as CompletionModel, M as Message } from '../types-D9rKQ_as.js';
|
|
2
2
|
import { Z as ZodSchema } from '../zod-schema-C7F4clpm.js';
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import { E as EmbeddingModel } from '../types-
|
|
3
|
+
import { a as AgentInteractionRequest, b as AgentInteractionResponse } from '../interactions-csssHYcN.js';
|
|
4
|
+
import { t as AgentSuspendedResult, j as AgentResponse, l as AgentRunOptions, k as AgentResult } from '../types-CElwurHZ.js';
|
|
5
|
+
import { E as EmbeddingModel } from '../types-Cr4uiYo5.js';
|
|
6
|
+
import '../model-call-options-CZkSw_xN.js';
|
|
6
7
|
import 'zod';
|
|
7
|
-
import '../guardrails/index.js';
|
|
8
8
|
import '../type-utils-CtHVDRn_.js';
|
|
9
|
-
import '../
|
|
10
|
-
import '../
|
|
11
|
-
import '../
|
|
12
|
-
import '../
|
|
13
|
-
import '@modelcontextprotocol/sdk/client/sse.js';
|
|
14
|
-
import '@modelcontextprotocol/sdk/client/stdio.js';
|
|
15
|
-
import '@modelcontextprotocol/sdk/client/streamableHttp.js';
|
|
16
|
-
import '../types-DhQfwiZn.js';
|
|
17
|
-
import '../dynamic-tools-i9woysFa.js';
|
|
18
|
-
import '../types-DT3nEemY.js';
|
|
19
|
-
import '../retry-D3Ruy-ba.js';
|
|
9
|
+
import '../retry-CjvSlKGW.js';
|
|
10
|
+
import '../types-CIPaIbZo.js';
|
|
11
|
+
import '../types-DbolvLtZ.js';
|
|
12
|
+
import '../tool-r62gf3QY.js';
|
|
20
13
|
|
|
21
14
|
type EvalOutcome<Score = unknown> = {
|
|
22
15
|
outcome: "pass";
|
|
@@ -57,7 +50,8 @@ declare const EvalOutcome: {
|
|
|
57
50
|
}): EvalOutcome<Score>;
|
|
58
51
|
};
|
|
59
52
|
|
|
60
|
-
type EvalMetadata =
|
|
53
|
+
type EvalMetadata = JsonObject;
|
|
54
|
+
type EvalReporterErrorPolicy = "collect" | "throw";
|
|
61
55
|
type EvalRunOptions = {
|
|
62
56
|
id?: string | undefined;
|
|
63
57
|
datasetName?: string | undefined;
|
|
@@ -85,6 +79,7 @@ type EvalTurn = {
|
|
|
85
79
|
metadata?: EvalMetadata | undefined;
|
|
86
80
|
};
|
|
87
81
|
type EvalTraceRef = {
|
|
82
|
+
observer?: string | undefined;
|
|
88
83
|
traceId: string;
|
|
89
84
|
observationId?: string | undefined;
|
|
90
85
|
responseId?: string | undefined;
|
|
@@ -255,7 +250,7 @@ type RunEvalSuiteOptions<Input, Output, Expected = unknown, Metrics extends read
|
|
|
255
250
|
concurrency?: number | undefined;
|
|
256
251
|
trace?: EvalTraceSelector<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
|
|
257
252
|
reporters?: readonly EvalReporter<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>>[] | undefined;
|
|
258
|
-
|
|
253
|
+
reporterErrorPolicy?: EvalReporterErrorPolicy | undefined;
|
|
259
254
|
targetUsage?: EvalTargetUsageSelector<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
|
|
260
255
|
cost?: EvalCostOptions<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
|
|
261
256
|
};
|
|
@@ -388,12 +383,41 @@ declare function knowledgeRetention<Input, Output, Expected = unknown, const Nam
|
|
|
388
383
|
}): EvalMetric<Input, Output, number, Expected, Name>;
|
|
389
384
|
type ConversationSource = EvalTurn[] | Message[];
|
|
390
385
|
|
|
391
|
-
type
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
386
|
+
type EvaluableAgent<Output> = {
|
|
387
|
+
generate(input: AgentRunOptions<Output>): Promise<AgentResult<Output>>;
|
|
388
|
+
};
|
|
389
|
+
type IsAny<Value> = 0 extends 1 & Value ? true : false;
|
|
390
|
+
type IsExactly<Left, Right> = IsAny<Left> extends true ? false : [Left] extends [Right] ? [Right] extends [Left] ? true : false : false;
|
|
391
|
+
type AgentEvalTargetOptions<Input, AgentOutput = string, Output = AgentResponse<AgentOutput>, Expected = unknown> = {
|
|
392
|
+
agent: EvaluableAgent<AgentOutput>;
|
|
393
|
+
request(args: {
|
|
394
|
+
input: Input;
|
|
395
|
+
testCase: EvalCase<Input, Expected>;
|
|
396
|
+
}): AgentRunOptions<AgentOutput> | Promise<AgentRunOptions<AgentOutput>>;
|
|
397
|
+
interactions?: {
|
|
398
|
+
maxResponses?: number | undefined;
|
|
399
|
+
respond(args: {
|
|
400
|
+
interaction: AgentInteractionRequest;
|
|
401
|
+
testCase: EvalCase<Input, Expected>;
|
|
402
|
+
phase: number;
|
|
403
|
+
}): AgentInteractionResponse | Promise<AgentInteractionResponse>;
|
|
404
|
+
} | undefined;
|
|
405
|
+
} & (IsExactly<Output, AgentResponse<AgentOutput>> extends true ? {
|
|
406
|
+
output?(args: {
|
|
407
|
+
response: AgentResponse<AgentOutput>;
|
|
408
|
+
testCase: EvalCase<Input, Expected>;
|
|
409
|
+
}): Output | Promise<Output>;
|
|
410
|
+
} : {
|
|
411
|
+
output(args: {
|
|
412
|
+
response: AgentResponse<AgentOutput>;
|
|
413
|
+
testCase: EvalCase<Input, Expected>;
|
|
414
|
+
}): Output | Promise<Output>;
|
|
415
|
+
});
|
|
416
|
+
declare class AgentEvalSuspensionError extends Error {
|
|
417
|
+
readonly result: AgentSuspendedResult;
|
|
418
|
+
constructor(result: AgentSuspendedResult, message?: string);
|
|
419
|
+
}
|
|
420
|
+
declare function agentEvalTarget<Input, AgentOutput = string, Output = AgentResponse<AgentOutput>, Expected = unknown>(options: AgentEvalTargetOptions<Input, AgentOutput, Output, Expected>): EvalTarget<Input, Output, Expected>;
|
|
397
421
|
|
|
398
422
|
type EvalOutputFormat = "pretty" | "json" | "quiet";
|
|
399
423
|
type EvalExpectedTotals = Partial<EvalTotals> & {
|
|
@@ -569,6 +593,10 @@ declare function resolveEvalTraceRef(args: {
|
|
|
569
593
|
}): EvalTraceRef | undefined;
|
|
570
594
|
declare function defaultEvalTraceSelector<Input, Output, Expected>(args: EvalTraceSelectorArgs<Input, Output, Expected>): EvalTraceRef | undefined;
|
|
571
595
|
|
|
596
|
+
declare class EvalReporterDispatchError extends AggregateError {
|
|
597
|
+
readonly phase: string;
|
|
598
|
+
constructor(phase: string, errors: readonly unknown[]);
|
|
599
|
+
}
|
|
572
600
|
declare function runEvalSuite<Input, Output, Expected = unknown, const Metrics extends readonly EvalMetric<NoInfer<Input>, NoInfer<Output>, unknown, NoInfer<Expected>, string>[] = readonly EvalMetric<NoInfer<Input>, NoInfer<Output>, unknown, NoInfer<Expected>, string>[]>(options: RunEvalSuiteOptions<Input, Output, Expected, Metrics>): Promise<EvalSuiteResult<Input, Output, Expected, Metrics>>;
|
|
573
601
|
|
|
574
602
|
declare function selectPromptOutput(args: EvalMetricArgs<unknown, unknown, unknown>): string;
|
|
@@ -640,4 +668,4 @@ declare function defineEvalSuite<const Cases extends readonly EvalCaseLike[], co
|
|
|
640
668
|
target: Target;
|
|
641
669
|
};
|
|
642
670
|
|
|
643
|
-
export { type AbstentionCategory, type AbstentionOptions, type AgentEvalTargetOptions, type AnswerRelevancyOptions, type AnyEvalMetric, type ContainsAllOptions, type ContainsAnyOptions, type ContainsListOptions, type ContainsOptions, type DefaultEvalActual, type DefinedEvalSuite, type DoesNotMatchOptions, EvalAssertionError, type EvalCase, type EvalCaseRequirements, type EvalCaseResult, type EvalCasesExpected, type EvalCasesForMetrics, type EvalCasesInput, type EvalCostCalculatorArgs, type EvalCostOptions, type EvalCostSummary, type EvalDataType, type EvalExpectations, type EvalExpectedOutcomes, type EvalExpectedTotals, type EvalMetadata, type EvalMetric, type EvalMetricArgs, type EvalMetricDescriptor, type EvalMetricResult, type EvalMetricResultFor, type EvalMetricScore, EvalOutcome, type EvalOutcomeStatus, type EvalOutputFormat, type EvalOutputWriters, type EvalReportArgs, type EvalReporter, type EvalRunContext, type EvalRunEndArgs, type EvalRunOptions, type EvalRunStartArgs, type EvalScoreDirection, type EvalScoreMap, type EvalScoreProjection, type EvalSuiteResult, type EvalTarget, type EvalTargetUsageSelector, type EvalTotals, type EvalTraceCarrier, type EvalTraceRef, type EvalTraceSelector, type EvalTraceSelectorArgs, type EvalTurn, type EvalUsageSummary, type ExactMatchOptions, type FaithfulnessOptions, type GEvalOptions, type GEvalParameter, type GEvalRubric, type HallucinationOptions, type JsonCorrectnessOptions, type KnowledgeRetentionOptions, type LlmJudgeOptions, type LlmScoreMetricScore, type LlmScoreOptions, type MatchesOptions, type MaxLengthOptions, type NotContainsOptions, type PrintEvalResultOptions, type PromptAlignmentOptions, type RequiredFieldsOptions, type RunEvalCliOptions, type RunEvalSuiteOptions, type SelectorOrValue, type SemanticSimilarityOptions, type SummarizationOptions, type TurnRelevancyOptions, type ValueSelector, abstention, agentEvalTarget, answerRelevancy, assertEvalOutcomes, assertEvalTotals, contains, containsAll, containsAny, defaultEvalTraceSelector, defineEvalCases, defineEvalSuite, defineMetric, doesNotMatch, evalExitCode, exactMatch, faithfulness, gEval, hallucination, jsonCorrectness, knowledgeRetention, llmJudge, llmScore, matches, maxLength, notContains, printEvalResult, projectEvalOutcome, promptAlignment, requiredFields, resolveEvalTraceRef, runEvalCli, runEvalSuite, selectPromptOutput, semanticSimilarity, summarization, turnRelevancy };
|
|
671
|
+
export { type AbstentionCategory, type AbstentionOptions, AgentEvalSuspensionError, type AgentEvalTargetOptions, type AnswerRelevancyOptions, type AnyEvalMetric, type ContainsAllOptions, type ContainsAnyOptions, type ContainsListOptions, type ContainsOptions, type DefaultEvalActual, type DefinedEvalSuite, type DoesNotMatchOptions, EvalAssertionError, type EvalCase, type EvalCaseRequirements, type EvalCaseResult, type EvalCasesExpected, type EvalCasesForMetrics, type EvalCasesInput, type EvalCostCalculatorArgs, type EvalCostOptions, type EvalCostSummary, type EvalDataType, type EvalExpectations, type EvalExpectedOutcomes, type EvalExpectedTotals, type EvalMetadata, type EvalMetric, type EvalMetricArgs, type EvalMetricDescriptor, type EvalMetricResult, type EvalMetricResultFor, type EvalMetricScore, EvalOutcome, type EvalOutcomeStatus, type EvalOutputFormat, type EvalOutputWriters, type EvalReportArgs, type EvalReporter, EvalReporterDispatchError, type EvalReporterErrorPolicy, type EvalRunContext, type EvalRunEndArgs, type EvalRunOptions, type EvalRunStartArgs, type EvalScoreDirection, type EvalScoreMap, type EvalScoreProjection, type EvalSuiteResult, type EvalTarget, type EvalTargetUsageSelector, type EvalTotals, type EvalTraceCarrier, type EvalTraceRef, type EvalTraceSelector, type EvalTraceSelectorArgs, type EvalTurn, type EvalUsageSummary, type ExactMatchOptions, type FaithfulnessOptions, type GEvalOptions, type GEvalParameter, type GEvalRubric, type HallucinationOptions, type JsonCorrectnessOptions, type KnowledgeRetentionOptions, type LlmJudgeOptions, type LlmScoreMetricScore, type LlmScoreOptions, type MatchesOptions, type MaxLengthOptions, type NotContainsOptions, type PrintEvalResultOptions, type PromptAlignmentOptions, type RequiredFieldsOptions, type RunEvalCliOptions, type RunEvalSuiteOptions, type SelectorOrValue, type SemanticSimilarityOptions, type SummarizationOptions, type TurnRelevancyOptions, type ValueSelector, abstention, agentEvalTarget, answerRelevancy, assertEvalOutcomes, assertEvalTotals, contains, containsAll, containsAny, defaultEvalTraceSelector, defineEvalCases, defineEvalSuite, defineMetric, doesNotMatch, evalExitCode, exactMatch, faithfulness, gEval, hallucination, jsonCorrectness, knowledgeRetention, llmJudge, llmScore, matches, maxLength, notContains, printEvalResult, projectEvalOutcome, promptAlignment, requiredFields, resolveEvalTraceRef, runEvalCli, runEvalSuite, selectPromptOutput, semanticSimilarity, summarization, turnRelevancy };
|
package/dist/evals/index.js
CHANGED
|
@@ -1,33 +1,29 @@
|
|
|
1
1
|
import {
|
|
2
|
-
|
|
3
|
-
} from "../chunk-
|
|
4
|
-
import "../chunk-YK4WAAS4.js";
|
|
5
|
-
import "../chunk-6WGGTJE6.js";
|
|
6
|
-
import "../chunk-R2LJSQUW.js";
|
|
7
|
-
import "../chunk-5XKDW36Z.js";
|
|
2
|
+
AgentRunBlockedError
|
|
3
|
+
} from "../chunk-TRQ3XRFD.js";
|
|
8
4
|
import {
|
|
9
5
|
cosineSimilarity,
|
|
10
|
-
embedText
|
|
6
|
+
embedText
|
|
7
|
+
} from "../chunk-GQL5KRAH.js";
|
|
8
|
+
import {
|
|
11
9
|
mapWithConcurrency
|
|
12
|
-
} from "../chunk-
|
|
10
|
+
} from "../chunk-3RM57ZT2.js";
|
|
13
11
|
import {
|
|
14
|
-
|
|
15
|
-
} from "../chunk-
|
|
16
|
-
import "../chunk-
|
|
17
|
-
import "../chunk-M3IOWB4Y.js";
|
|
18
|
-
import "../chunk-RFDVKCIE.js";
|
|
12
|
+
extract
|
|
13
|
+
} from "../chunk-ZCPOIDAQ.js";
|
|
14
|
+
import "../chunk-MGNTXGTX.js";
|
|
19
15
|
import {
|
|
20
16
|
Usage
|
|
21
|
-
} from "../chunk-
|
|
22
|
-
import "../chunk-
|
|
23
|
-
import "../chunk-
|
|
17
|
+
} from "../chunk-D2ECOWN5.js";
|
|
18
|
+
import "../chunk-PK5LKOFL.js";
|
|
19
|
+
import "../chunk-FN7HLLAZ.js";
|
|
24
20
|
|
|
25
21
|
// src/evals/advanced-metrics.ts
|
|
26
22
|
import { z } from "zod";
|
|
27
23
|
|
|
28
24
|
// src/evals/format.ts
|
|
29
25
|
function defaultOutputValue(output) {
|
|
30
|
-
if (typeof output === "object" && output !== null && "output" in output
|
|
26
|
+
if (typeof output === "object" && output !== null && "output" in output) {
|
|
31
27
|
return output.output;
|
|
32
28
|
}
|
|
33
29
|
return output;
|
|
@@ -54,16 +50,15 @@ function errorMessage(error) {
|
|
|
54
50
|
|
|
55
51
|
// src/evals/judge.ts
|
|
56
52
|
async function runJudge(args) {
|
|
57
|
-
const
|
|
53
|
+
const result = await extract({
|
|
58
54
|
model: args.model,
|
|
59
55
|
outputSchema: args.schema,
|
|
60
|
-
instructions: args.instructions
|
|
61
|
-
|
|
62
|
-
const result = await extractor.extractResult(args.prompt, {
|
|
56
|
+
instructions: args.instructions,
|
|
57
|
+
text: args.prompt,
|
|
63
58
|
temperature: 0,
|
|
64
59
|
retries: args.retries <= 0 ? void 0 : { maxAttempts: Math.trunc(args.retries) + 1 }
|
|
65
60
|
});
|
|
66
|
-
return { data: result.
|
|
61
|
+
return { data: result.output, usage: result.usage };
|
|
67
62
|
}
|
|
68
63
|
function addUsage(...values) {
|
|
69
64
|
return values.reduce((total, usage) => Usage.add(total, usage), Usage.empty());
|
|
@@ -1120,17 +1115,61 @@ function contentText(content) {
|
|
|
1120
1115
|
}
|
|
1121
1116
|
|
|
1122
1117
|
// src/evals/agent-target.ts
|
|
1123
|
-
|
|
1118
|
+
var AgentEvalSuspensionError = class extends Error {
|
|
1119
|
+
constructor(result, message = "Agent eval target suspended without an interaction responder.") {
|
|
1120
|
+
super(message);
|
|
1121
|
+
this.result = result;
|
|
1122
|
+
this.name = "AgentEvalSuspensionError";
|
|
1123
|
+
}
|
|
1124
|
+
result;
|
|
1125
|
+
};
|
|
1126
|
+
function agentEvalTarget(options) {
|
|
1124
1127
|
return async (input, testCase) => {
|
|
1125
|
-
const
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
|
|
1128
|
+
const maxResponses = options.interactions?.maxResponses ?? 10;
|
|
1129
|
+
if (!Number.isSafeInteger(maxResponses) || maxResponses < 1) {
|
|
1130
|
+
throw new TypeError("Agent eval interactions.maxResponses must be a positive integer.");
|
|
1131
|
+
}
|
|
1132
|
+
const request = await options.request({ input, testCase });
|
|
1133
|
+
const runSettings = agentRunSettings(request);
|
|
1134
|
+
let response = await options.agent.generate(request);
|
|
1135
|
+
let phase = 0;
|
|
1136
|
+
while (response.status === "suspended") {
|
|
1137
|
+
if (options.interactions === void 0) {
|
|
1138
|
+
throw new AgentEvalSuspensionError(response);
|
|
1139
|
+
}
|
|
1140
|
+
if (phase >= maxResponses) {
|
|
1141
|
+
throw new AgentEvalSuspensionError(
|
|
1142
|
+
response,
|
|
1143
|
+
`Agent eval target exceeded the interaction response limit of ${maxResponses}.`
|
|
1144
|
+
);
|
|
1145
|
+
}
|
|
1146
|
+
phase += 1;
|
|
1147
|
+
const interactionResponse = await options.interactions.respond({
|
|
1148
|
+
interaction: response.interaction,
|
|
1149
|
+
testCase,
|
|
1150
|
+
phase
|
|
1151
|
+
});
|
|
1152
|
+
response = await options.agent.generate({
|
|
1153
|
+
continuation: response.continuation,
|
|
1154
|
+
response: interactionResponse,
|
|
1155
|
+
...runSettings
|
|
1156
|
+
});
|
|
1130
1157
|
}
|
|
1131
|
-
|
|
1158
|
+
if (response.status === "blocked") throw new AgentRunBlockedError(response);
|
|
1159
|
+
return options.output === void 0 ? response : await options.output({ response, testCase });
|
|
1132
1160
|
};
|
|
1133
1161
|
}
|
|
1162
|
+
function agentRunSettings(request) {
|
|
1163
|
+
const {
|
|
1164
|
+
prompt: _prompt,
|
|
1165
|
+
messages: _messages,
|
|
1166
|
+
session: _session,
|
|
1167
|
+
continuation: _continuation,
|
|
1168
|
+
response: _response,
|
|
1169
|
+
...settings
|
|
1170
|
+
} = request;
|
|
1171
|
+
return settings;
|
|
1172
|
+
}
|
|
1134
1173
|
|
|
1135
1174
|
// src/evals/reporting.ts
|
|
1136
1175
|
function projectEvalOutcome(outcome, dataType, projectScore) {
|
|
@@ -1193,6 +1232,7 @@ function traceFromCarrier(value) {
|
|
|
1193
1232
|
function traceFromMetadata(metadata) {
|
|
1194
1233
|
if (metadata === void 0) return void 0;
|
|
1195
1234
|
return readTraceRef({
|
|
1235
|
+
observer: metadata.traceObserver,
|
|
1196
1236
|
traceId: metadata.traceId,
|
|
1197
1237
|
observationId: metadata.observationId,
|
|
1198
1238
|
responseId: metadata.responseId
|
|
@@ -1202,9 +1242,11 @@ function readTraceRef(value) {
|
|
|
1202
1242
|
if (typeof value !== "object" || value === null) return void 0;
|
|
1203
1243
|
const traceId = value.traceId;
|
|
1204
1244
|
if (typeof traceId !== "string" || traceId.length === 0) return void 0;
|
|
1245
|
+
const observer = value.observer;
|
|
1205
1246
|
const observationId = value.observationId;
|
|
1206
1247
|
const responseId = value.responseId;
|
|
1207
1248
|
const trace = { traceId };
|
|
1249
|
+
if (typeof observer === "string" && observer.length > 0) trace.observer = observer;
|
|
1208
1250
|
if (typeof observationId === "string" && observationId.length > 0) {
|
|
1209
1251
|
trace.observationId = observationId;
|
|
1210
1252
|
}
|
|
@@ -1213,6 +1255,14 @@ function readTraceRef(value) {
|
|
|
1213
1255
|
}
|
|
1214
1256
|
|
|
1215
1257
|
// src/evals/runner.ts
|
|
1258
|
+
var EvalReporterDispatchError = class extends AggregateError {
|
|
1259
|
+
phase;
|
|
1260
|
+
constructor(phase, errors) {
|
|
1261
|
+
super(errors, `Evaluation reporter ${phase} failed ${errors.length} time(s).`);
|
|
1262
|
+
this.name = "EvalReporterDispatchError";
|
|
1263
|
+
this.phase = phase;
|
|
1264
|
+
}
|
|
1265
|
+
};
|
|
1216
1266
|
async function runEvalSuite(options) {
|
|
1217
1267
|
validateSuiteOptions(options);
|
|
1218
1268
|
const startedAtMs = Date.now();
|
|
@@ -1229,7 +1279,7 @@ async function runEvalSuite(options) {
|
|
|
1229
1279
|
reporterErrors = await notifyRunStart(
|
|
1230
1280
|
reporters,
|
|
1231
1281
|
lifecycle,
|
|
1232
|
-
options.
|
|
1282
|
+
options.reporterErrorPolicy ?? "collect"
|
|
1233
1283
|
);
|
|
1234
1284
|
} catch (error) {
|
|
1235
1285
|
await notifyRunEnd(reporters, {
|
|
@@ -1264,25 +1314,26 @@ async function runEvalSuite(options) {
|
|
|
1264
1314
|
metrics: aggregates.metrics,
|
|
1265
1315
|
cases: aggregates.cases,
|
|
1266
1316
|
usage: aggregates.usage,
|
|
1267
|
-
...aggregates.cost === void 0 ? {} : { cost: aggregates.cost },
|
|
1268
1317
|
durationMs: Date.now() - startedAtMs,
|
|
1269
1318
|
reporterErrors
|
|
1270
1319
|
};
|
|
1320
|
+
if (aggregates.cost !== void 0) {
|
|
1321
|
+
result.cost = aggregates.cost;
|
|
1322
|
+
}
|
|
1323
|
+
const runEndArgs = {
|
|
1324
|
+
...lifecycle,
|
|
1325
|
+
status: "completed",
|
|
1326
|
+
completedAt,
|
|
1327
|
+
durationMs: result.durationMs,
|
|
1328
|
+
metrics: result.metrics,
|
|
1329
|
+
cases: result.cases,
|
|
1330
|
+
usage: result.usage
|
|
1331
|
+
};
|
|
1332
|
+
if (result.cost !== void 0) {
|
|
1333
|
+
runEndArgs.cost = result.cost;
|
|
1334
|
+
}
|
|
1271
1335
|
result.reporterErrors.push(
|
|
1272
|
-
...await notifyRunEnd(
|
|
1273
|
-
reporters,
|
|
1274
|
-
{
|
|
1275
|
-
...lifecycle,
|
|
1276
|
-
status: "completed",
|
|
1277
|
-
completedAt,
|
|
1278
|
-
durationMs: result.durationMs,
|
|
1279
|
-
metrics: result.metrics,
|
|
1280
|
-
cases: result.cases,
|
|
1281
|
-
usage: result.usage,
|
|
1282
|
-
...result.cost === void 0 ? {} : { cost: result.cost }
|
|
1283
|
-
},
|
|
1284
|
-
options.failOnReporterError === true
|
|
1285
|
-
)
|
|
1336
|
+
...await notifyRunEnd(reporters, runEndArgs, options.reporterErrorPolicy ?? "collect")
|
|
1286
1337
|
);
|
|
1287
1338
|
return result;
|
|
1288
1339
|
}
|
|
@@ -1335,7 +1386,7 @@ async function runEvalCase(options, testCase, run) {
|
|
|
1335
1386
|
trace: traceResult.trace,
|
|
1336
1387
|
traceError: traceResult.error,
|
|
1337
1388
|
reporters: options.reporters ?? [],
|
|
1338
|
-
|
|
1389
|
+
reporterErrorPolicy: options.reporterErrorPolicy ?? "collect"
|
|
1339
1390
|
});
|
|
1340
1391
|
const metricResult = {
|
|
1341
1392
|
metricName: metric.name,
|
|
@@ -1388,9 +1439,7 @@ async function safeEvaluate(suiteName, testCase, output, metric) {
|
|
|
1388
1439
|
async function reportOutcome(args) {
|
|
1389
1440
|
const errors = [];
|
|
1390
1441
|
if (args.traceError !== void 0) {
|
|
1391
|
-
if (args.failOnReporterError) throw args.traceError;
|
|
1392
1442
|
errors.push(args.traceError);
|
|
1393
|
-
return errors;
|
|
1394
1443
|
}
|
|
1395
1444
|
for (const reporter of args.reporters) {
|
|
1396
1445
|
try {
|
|
@@ -1405,12 +1454,10 @@ async function reportOutcome(args) {
|
|
|
1405
1454
|
outcome: args.outcome
|
|
1406
1455
|
});
|
|
1407
1456
|
} catch (error) {
|
|
1408
|
-
if (args.failOnReporterError) {
|
|
1409
|
-
throw error;
|
|
1410
|
-
}
|
|
1411
1457
|
errors.push(error);
|
|
1412
1458
|
}
|
|
1413
1459
|
}
|
|
1460
|
+
throwReporterErrors("report", errors, args.reporterErrorPolicy);
|
|
1414
1461
|
return errors;
|
|
1415
1462
|
}
|
|
1416
1463
|
function resolveRun(options, startedAtMs) {
|
|
@@ -1426,40 +1473,46 @@ function resolveRun(options, startedAtMs) {
|
|
|
1426
1473
|
throw new TypeError(`Evaluation run ${label} must contain 1 to 256 characters`);
|
|
1427
1474
|
}
|
|
1428
1475
|
}
|
|
1429
|
-
|
|
1476
|
+
const run = {
|
|
1430
1477
|
id,
|
|
1431
|
-
startedAt: new Date(startedAtMs).toISOString()
|
|
1432
|
-
...options.run?.datasetName === void 0 ? {} : { datasetName: options.run.datasetName },
|
|
1433
|
-
...options.run?.datasetVersion === void 0 ? {} : { datasetVersion: options.run.datasetVersion },
|
|
1434
|
-
...options.run?.metadata === void 0 ? {} : { metadata: options.run.metadata }
|
|
1478
|
+
startedAt: new Date(startedAtMs).toISOString()
|
|
1435
1479
|
};
|
|
1480
|
+
if (options.run?.datasetName !== void 0) run.datasetName = options.run.datasetName;
|
|
1481
|
+
if (options.run?.datasetVersion !== void 0) run.datasetVersion = options.run.datasetVersion;
|
|
1482
|
+
if (options.run?.metadata !== void 0) run.metadata = options.run.metadata;
|
|
1483
|
+
return run;
|
|
1436
1484
|
}
|
|
1437
|
-
async function notifyRunStart(reporters, args,
|
|
1485
|
+
async function notifyRunStart(reporters, args, errorPolicy) {
|
|
1438
1486
|
const errors = [];
|
|
1439
1487
|
for (const reporter of reporters) {
|
|
1440
1488
|
if (reporter.onRunStart === void 0) continue;
|
|
1441
1489
|
try {
|
|
1442
1490
|
await reporter.onRunStart(args);
|
|
1443
1491
|
} catch (error) {
|
|
1444
|
-
if (failOnReporterError) throw error;
|
|
1445
1492
|
errors.push(error);
|
|
1446
1493
|
}
|
|
1447
1494
|
}
|
|
1495
|
+
throwReporterErrors("onRunStart", errors, errorPolicy);
|
|
1448
1496
|
return errors;
|
|
1449
1497
|
}
|
|
1450
|
-
async function notifyRunEnd(reporters, args,
|
|
1498
|
+
async function notifyRunEnd(reporters, args, errorPolicy = "collect") {
|
|
1451
1499
|
const errors = [];
|
|
1452
1500
|
for (const reporter of reporters) {
|
|
1453
1501
|
if (reporter.onRunEnd === void 0) continue;
|
|
1454
1502
|
try {
|
|
1455
1503
|
await reporter.onRunEnd(args);
|
|
1456
1504
|
} catch (error) {
|
|
1457
|
-
if (failOnReporterError) throw error;
|
|
1458
1505
|
errors.push(error);
|
|
1459
1506
|
}
|
|
1460
1507
|
}
|
|
1508
|
+
throwReporterErrors("onRunEnd", errors, errorPolicy);
|
|
1461
1509
|
return errors;
|
|
1462
1510
|
}
|
|
1511
|
+
function throwReporterErrors(phase, errors, errorPolicy) {
|
|
1512
|
+
if (errorPolicy === "throw" && errors.length > 0) {
|
|
1513
|
+
throw new EvalReporterDispatchError(phase, errors);
|
|
1514
|
+
}
|
|
1515
|
+
}
|
|
1463
1516
|
function countMetricOutcomes(results) {
|
|
1464
1517
|
const totals = emptyTotals();
|
|
1465
1518
|
for (const result of results) {
|
|
@@ -1714,18 +1767,21 @@ function jsonResult(result) {
|
|
|
1714
1767
|
}
|
|
1715
1768
|
function expectationMismatches(result, expectations) {
|
|
1716
1769
|
if (expectations === void 0) return [];
|
|
1717
|
-
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1770
|
+
const mismatches = [];
|
|
1771
|
+
if (expectations.totals !== void 0) {
|
|
1772
|
+
mismatches.push(...totalMismatches(result, expectations.totals));
|
|
1773
|
+
}
|
|
1774
|
+
if (expectations.outcomes !== void 0) {
|
|
1775
|
+
mismatches.push(...outcomeMismatches(result, expectations.outcomes));
|
|
1776
|
+
}
|
|
1777
|
+
return mismatches;
|
|
1721
1778
|
}
|
|
1722
1779
|
function totalMismatches(result, expected) {
|
|
1723
|
-
const directMetrics = {
|
|
1724
|
-
|
|
1725
|
-
|
|
1726
|
-
|
|
1727
|
-
|
|
1728
|
-
};
|
|
1780
|
+
const directMetrics = {};
|
|
1781
|
+
if (expected.total !== void 0) directMetrics.total = expected.total;
|
|
1782
|
+
if (expected.passed !== void 0) directMetrics.passed = expected.passed;
|
|
1783
|
+
if (expected.failed !== void 0) directMetrics.failed = expected.failed;
|
|
1784
|
+
if (expected.invalid !== void 0) directMetrics.invalid = expected.invalid;
|
|
1729
1785
|
return [
|
|
1730
1786
|
...totalsGroupMismatches("metrics", result.metrics, {
|
|
1731
1787
|
...directMetrics,
|
|
@@ -2004,9 +2060,9 @@ function semanticSimilarity(options) {
|
|
|
2004
2060
|
if (typeof expected !== "string") {
|
|
2005
2061
|
return EvalOutcome.invalid("Semantic similarity expected value must be a string.");
|
|
2006
2062
|
}
|
|
2007
|
-
const [actualEmbedding, expectedEmbedding] = await Promise.all([
|
|
2008
|
-
embedText(options.model, actual),
|
|
2009
|
-
embedText(options.model, expected)
|
|
2063
|
+
const [{ embedding: actualEmbedding }, { embedding: expectedEmbedding }] = await Promise.all([
|
|
2064
|
+
embedText({ model: options.model, text: actual }),
|
|
2065
|
+
embedText({ model: options.model, text: expected })
|
|
2010
2066
|
]);
|
|
2011
2067
|
const score = cosineSimilarity(actualEmbedding.vector, expectedEmbedding.vector);
|
|
2012
2068
|
return score >= options.threshold ? EvalOutcome.pass(score) : EvalOutcome.fail(score, { comment: `Similarity below threshold ${options.threshold}.` });
|
|
@@ -2014,23 +2070,19 @@ function semanticSimilarity(options) {
|
|
|
2014
2070
|
};
|
|
2015
2071
|
}
|
|
2016
2072
|
function llmJudge(options) {
|
|
2017
|
-
const extractor = new Extractor({
|
|
2018
|
-
model: options.model,
|
|
2019
|
-
outputSchema: options.schema,
|
|
2020
|
-
instructions: options.instructions ?? "Judge the eval case by the requested schema. Submit the judgment using the schema."
|
|
2021
|
-
});
|
|
2022
2073
|
return {
|
|
2023
2074
|
name: options.name ?? "llm_judge",
|
|
2024
2075
|
required: options.required ?? true,
|
|
2025
2076
|
async evaluate(args) {
|
|
2026
2077
|
try {
|
|
2027
|
-
const result = await
|
|
2028
|
-
|
|
2029
|
-
|
|
2030
|
-
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2078
|
+
const result = await extract({
|
|
2079
|
+
model: options.model,
|
|
2080
|
+
outputSchema: options.schema,
|
|
2081
|
+
instructions: options.instructions ?? "Judge the eval case by the requested schema. Submit the judgment using the schema.",
|
|
2082
|
+
text: await resolveJudgePrompt(options.prompt, args),
|
|
2083
|
+
retries: evalExtractionRetries(options.retries)
|
|
2084
|
+
});
|
|
2085
|
+
return options.passes(result.output) ? EvalOutcome.pass(result.output, { usage: result.usage }) : EvalOutcome.fail(result.output, { usage: result.usage });
|
|
2034
2086
|
} catch (error) {
|
|
2035
2087
|
return EvalOutcome.invalid(errorMessage(error));
|
|
2036
2088
|
}
|
|
@@ -2039,17 +2091,6 @@ function llmJudge(options) {
|
|
|
2039
2091
|
}
|
|
2040
2092
|
function llmScore(options) {
|
|
2041
2093
|
const criteria = Array.isArray(options.criteria) ? options.criteria.join("\n") : options.criteria;
|
|
2042
|
-
const extractor = new Extractor({
|
|
2043
|
-
model: options.model,
|
|
2044
|
-
outputSchema: z2.object({
|
|
2045
|
-
score: z2.number(),
|
|
2046
|
-
feedback: z2.string()
|
|
2047
|
-
}),
|
|
2048
|
-
instructions: options.instructions ?? `Score the eval case against these criteria:
|
|
2049
|
-
${criteria}
|
|
2050
|
-
|
|
2051
|
-
Return a score between 0 and 1 and brief feedback.`
|
|
2052
|
-
});
|
|
2053
2094
|
return {
|
|
2054
2095
|
name: options.name ?? "llm_score",
|
|
2055
2096
|
required: options.required ?? true,
|
|
@@ -2059,13 +2100,20 @@ Return a score between 0 and 1 and brief feedback.`
|
|
|
2059
2100
|
threshold: options.threshold,
|
|
2060
2101
|
async evaluate(args) {
|
|
2061
2102
|
try {
|
|
2062
|
-
const result = await
|
|
2063
|
-
|
|
2064
|
-
{
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2068
|
-
|
|
2103
|
+
const result = await extract({
|
|
2104
|
+
model: options.model,
|
|
2105
|
+
outputSchema: z2.object({
|
|
2106
|
+
score: z2.number(),
|
|
2107
|
+
feedback: z2.string()
|
|
2108
|
+
}),
|
|
2109
|
+
instructions: options.instructions ?? `Score the eval case against these criteria:
|
|
2110
|
+
${criteria}
|
|
2111
|
+
|
|
2112
|
+
Return a score between 0 and 1 and brief feedback.`,
|
|
2113
|
+
text: await resolveJudgePrompt(options.prompt, args),
|
|
2114
|
+
retries: evalExtractionRetries(options.retries)
|
|
2115
|
+
});
|
|
2116
|
+
const score = result.output;
|
|
2069
2117
|
if (score.score < 0 || score.score > 1) {
|
|
2070
2118
|
return EvalOutcome.invalid(`Score ${score.score} outside valid range [0, 1].`, {
|
|
2071
2119
|
score,
|
|
@@ -2100,8 +2148,10 @@ function defineEvalSuite(options) {
|
|
|
2100
2148
|
};
|
|
2101
2149
|
}
|
|
2102
2150
|
export {
|
|
2151
|
+
AgentEvalSuspensionError,
|
|
2103
2152
|
EvalAssertionError,
|
|
2104
2153
|
EvalOutcome,
|
|
2154
|
+
EvalReporterDispatchError,
|
|
2105
2155
|
abstention,
|
|
2106
2156
|
agentEvalTarget,
|
|
2107
2157
|
answerRelevancy,
|