@tangle-network/agent-eval 0.115.3 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/dist/analyst/index.d.ts +16 -11
- package/dist/analyst/index.js +33 -25
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +12 -5
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +247 -34
- package/dist/campaign/index.js +33 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +45 -31
- package/dist/contract/index.js +58 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
- package/dist/hosted/index.d.ts +14 -7
- package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +97 -55
- package/dist/index.js +343 -244
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
- package/dist/kind-factory-ClZmO25A.d.ts +171 -0
- package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +10 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
- package/dist/rl.d.ts +17 -12
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +19 -10
- package/dist/traces.js +16 -4
- package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-5S5NJ63F.js.map +0 -1
- package/dist/chunk-ADYLPOSX.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-I6LVHOV3.js +0 -205
- package/dist/chunk-I6LVHOV3.js.map +0 -1
- package/dist/chunk-KG4TD7EQ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-QMXXSNC4.js +0 -761
- package/dist/chunk-QMXXSNC4.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/chunk-WSBUZMBU.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -9,12 +9,12 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-FQNLDL4D.js";
|
|
13
13
|
import {
|
|
14
14
|
BENCHMARK_SPLIT_SEED,
|
|
15
15
|
benchmarks_exports,
|
|
16
16
|
deterministicSplit
|
|
17
|
-
} from "./chunk-
|
|
17
|
+
} from "./chunk-JSDVRFAP.js";
|
|
18
18
|
import {
|
|
19
19
|
DEFAULT_RULES,
|
|
20
20
|
buildTrajectory,
|
|
@@ -23,7 +23,7 @@ import {
|
|
|
23
23
|
computeToolUseMetrics,
|
|
24
24
|
iqr,
|
|
25
25
|
welchsTTest
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-ODVOOEWQ.js";
|
|
27
27
|
import {
|
|
28
28
|
analyzeSeries
|
|
29
29
|
} from "./chunk-BOD4O7OF.js";
|
|
@@ -40,40 +40,45 @@ import {
|
|
|
40
40
|
import {
|
|
41
41
|
CODING_HARNESSES,
|
|
42
42
|
HARNESS_NATIVE_MODEL,
|
|
43
|
-
JudgeParseError,
|
|
44
|
-
adversarialJudge,
|
|
45
43
|
agentProfileHash,
|
|
46
44
|
agentProfileId,
|
|
47
45
|
agentProfileModelId,
|
|
48
|
-
codeExecutionJudge,
|
|
49
|
-
coherenceJudge,
|
|
50
46
|
comparePairedArms,
|
|
51
47
|
completionVerdict,
|
|
52
|
-
createCustomJudge,
|
|
53
|
-
createDomainExpertJudge,
|
|
54
48
|
createLlmCorrectnessChecker,
|
|
55
49
|
createTokenRecallChecker,
|
|
56
|
-
defaultJudges,
|
|
57
50
|
expandProfileAxes,
|
|
58
51
|
extractProducedState,
|
|
59
52
|
harnessAxisOf,
|
|
60
|
-
llmJudge,
|
|
61
53
|
pairArms,
|
|
62
54
|
parseCorrectnessResponse,
|
|
63
55
|
verifyCompletion
|
|
64
|
-
} from "./chunk-
|
|
56
|
+
} from "./chunk-ZUXV7UWZ.js";
|
|
65
57
|
import {
|
|
66
58
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
67
59
|
DEFAULT_RED_TEAM_CORPUS,
|
|
68
60
|
Dataset,
|
|
69
61
|
HoldoutLockedError,
|
|
62
|
+
JudgeParseError,
|
|
63
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
64
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
65
|
+
adversarialJudge,
|
|
70
66
|
buildReflectionPrompt,
|
|
71
67
|
campaignMeanComposite,
|
|
68
|
+
codeExecutionJudge,
|
|
69
|
+
coherenceJudge,
|
|
70
|
+
costReceiptFromTCloud,
|
|
71
|
+
createCustomJudge,
|
|
72
|
+
createDomainExpertJudge,
|
|
73
|
+
createReferenceEquivalenceJudge,
|
|
72
74
|
crowdingDistance,
|
|
75
|
+
defaultJudges,
|
|
73
76
|
dominates,
|
|
74
77
|
gepaProposer,
|
|
75
78
|
hashScenarios,
|
|
76
79
|
heldOutGate,
|
|
80
|
+
llmJudge,
|
|
81
|
+
maximumChargeForTCloudRequest,
|
|
77
82
|
paretoFrontier,
|
|
78
83
|
paretoFrontierWithCrowding,
|
|
79
84
|
parseReflectionResponse,
|
|
@@ -81,20 +86,12 @@ import {
|
|
|
81
86
|
redTeamReport,
|
|
82
87
|
runCanaries,
|
|
83
88
|
runImprovementLoop,
|
|
89
|
+
runReferenceEquivalenceJudge,
|
|
84
90
|
scalarScore,
|
|
85
91
|
scoreRedTeamOutput,
|
|
86
92
|
surfaceContentHash,
|
|
87
93
|
toolNamesForRun
|
|
88
|
-
} from "./chunk-
|
|
89
|
-
import {
|
|
90
|
-
MODEL_PRICING,
|
|
91
|
-
MetricsCollector,
|
|
92
|
-
TokenCounter,
|
|
93
|
-
estimateCost,
|
|
94
|
-
estimateTokens,
|
|
95
|
-
isModelPriced,
|
|
96
|
-
resolveModelPricing
|
|
97
|
-
} from "./chunk-VI2UW6B6.js";
|
|
94
|
+
} from "./chunk-HQPHZGL6.js";
|
|
98
95
|
import {
|
|
99
96
|
BackendIntegrityError,
|
|
100
97
|
assertRealBackend,
|
|
@@ -104,7 +101,7 @@ import {
|
|
|
104
101
|
fileVerdictCache,
|
|
105
102
|
inMemoryVerdictCache,
|
|
106
103
|
summarizeBackendIntegrity
|
|
107
|
-
} from "./chunk-
|
|
104
|
+
} from "./chunk-IDZTTFRR.js";
|
|
108
105
|
import {
|
|
109
106
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
110
107
|
FindingsStore,
|
|
@@ -114,46 +111,50 @@ import {
|
|
|
114
111
|
SKILL_USAGE_ANALYST,
|
|
115
112
|
SkillUsageAnalyst,
|
|
116
113
|
createAnalystAi,
|
|
117
|
-
createChatClient,
|
|
118
114
|
createSemanticConceptJudge,
|
|
119
115
|
defaultIsMaterial,
|
|
120
116
|
diffFindings,
|
|
121
117
|
runSemanticConceptJudge
|
|
122
|
-
} from "./chunk-
|
|
118
|
+
} from "./chunk-CCZIVI3F.js";
|
|
123
119
|
import {
|
|
124
120
|
buildDefaultAnalystRegistry,
|
|
125
|
-
computeTraceMetrics
|
|
126
|
-
|
|
121
|
+
computeTraceMetrics,
|
|
122
|
+
createChatClient
|
|
123
|
+
} from "./chunk-VF3XSYTI.js";
|
|
124
|
+
import "./chunk-HHWE3POT.js";
|
|
125
|
+
import {
|
|
126
|
+
Mutex
|
|
127
|
+
} from "./chunk-3YYRZDON.js";
|
|
127
128
|
import {
|
|
129
|
+
AnalystRegistry,
|
|
128
130
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
129
|
-
|
|
131
|
+
DEFAULT_TRACE_ANALYST_KINDS,
|
|
132
|
+
FAILURE_MODE_KIND_SPEC,
|
|
133
|
+
IMPROVEMENT_KIND_SPEC,
|
|
134
|
+
KNOWLEDGE_GAP_KIND_SPEC,
|
|
135
|
+
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
130
136
|
POLICY_EDIT_AXES,
|
|
137
|
+
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
131
138
|
POLICY_EDIT_TARGET_SURFACES,
|
|
132
139
|
PolicyEditValidationError,
|
|
133
140
|
admitPolicyEdit,
|
|
134
141
|
aggregateRunScore,
|
|
135
142
|
applyPolicyEditToSurface,
|
|
136
143
|
clamp01,
|
|
144
|
+
computeFindingId,
|
|
137
145
|
computePolicyEditId,
|
|
146
|
+
createTraceAnalystKind,
|
|
138
147
|
isPolicyEdit,
|
|
148
|
+
makeFinding,
|
|
139
149
|
makePolicyEdit,
|
|
150
|
+
makePolicyEditCandidateRecord,
|
|
140
151
|
policyEditFromFinding,
|
|
141
152
|
policyEditsFromFindings,
|
|
153
|
+
renderPriorFindings,
|
|
142
154
|
scorePolicyEditReadiness,
|
|
143
|
-
validatePolicyEdit
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
AnalystRegistry,
|
|
147
|
-
DEFAULT_TRACE_ANALYST_KINDS,
|
|
148
|
-
FAILURE_MODE_KIND_SPEC,
|
|
149
|
-
IMPROVEMENT_KIND_SPEC,
|
|
150
|
-
KNOWLEDGE_GAP_KIND_SPEC,
|
|
151
|
-
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
152
|
-
computeFindingId,
|
|
153
|
-
createTraceAnalystKind,
|
|
154
|
-
makeFinding,
|
|
155
|
-
renderPriorFindings
|
|
156
|
-
} from "./chunk-5S5NJ63F.js";
|
|
155
|
+
validatePolicyEdit,
|
|
156
|
+
validatePolicyEditCandidateRecord
|
|
157
|
+
} from "./chunk-MGEHEHSN.js";
|
|
157
158
|
import {
|
|
158
159
|
allCriticalPassed,
|
|
159
160
|
controlFailureClassFromVerification,
|
|
@@ -184,7 +185,7 @@ import {
|
|
|
184
185
|
} from "./chunk-MOXWMGPC.js";
|
|
185
186
|
import {
|
|
186
187
|
runEvalCampaign
|
|
187
|
-
} from "./chunk-
|
|
188
|
+
} from "./chunk-GQCZRZ7L.js";
|
|
188
189
|
import "./chunk-ARU2PZFM.js";
|
|
189
190
|
import {
|
|
190
191
|
evaluateInterimReleaseConfidence,
|
|
@@ -267,13 +268,14 @@ import {
|
|
|
267
268
|
scoreTraceInsightReadiness,
|
|
268
269
|
tokenizeDomainWords,
|
|
269
270
|
traceAnalystOnRunComplete
|
|
270
|
-
} from "./chunk-
|
|
271
|
+
} from "./chunk-YZPO4UHR.js";
|
|
271
272
|
import {
|
|
272
273
|
FAILURE_CLASSES,
|
|
273
274
|
TRACE_SCHEMA_VERSION,
|
|
274
275
|
aggregateLlm,
|
|
275
276
|
argHash,
|
|
276
277
|
groupBy,
|
|
278
|
+
hasCapturedToolArgs,
|
|
277
279
|
isJudgeSpan,
|
|
278
280
|
isLlmSpan,
|
|
279
281
|
isRetrievalSpan,
|
|
@@ -284,13 +286,13 @@ import {
|
|
|
284
286
|
runFailureClass,
|
|
285
287
|
runsForScenario,
|
|
286
288
|
toolSpans
|
|
287
|
-
} from "./chunk-
|
|
289
|
+
} from "./chunk-LQUTGLOZ.js";
|
|
288
290
|
import {
|
|
289
291
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
290
292
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
291
293
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
292
294
|
analyzeTraces
|
|
293
|
-
} from "./chunk-
|
|
295
|
+
} from "./chunk-4JLWXDYA.js";
|
|
294
296
|
import {
|
|
295
297
|
DEFAULT_REDACTION_RULES,
|
|
296
298
|
REDACTION_VERSION,
|
|
@@ -299,6 +301,7 @@ import {
|
|
|
299
301
|
} from "./chunk-GGE4NNQT.js";
|
|
300
302
|
import {
|
|
301
303
|
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
304
|
+
INPUT_VALUE,
|
|
302
305
|
LLM_CACHED_TOKENS,
|
|
303
306
|
LLM_CACHED_TOKEN_ATTR_KEYS,
|
|
304
307
|
LLM_COST_ATTR_KEYS,
|
|
@@ -310,14 +313,18 @@ import {
|
|
|
310
313
|
LLM_OUTPUT_TOKENS,
|
|
311
314
|
LLM_OUTPUT_TOKEN_ATTR_KEYS,
|
|
312
315
|
OPENINFERENCE_SPAN_KIND,
|
|
316
|
+
OUTPUT_VALUE,
|
|
313
317
|
OtlpFileTraceStore,
|
|
314
318
|
SPAN_KIND_ATTR_KEYS,
|
|
315
319
|
SpanNotFoundError,
|
|
320
|
+
TOOL_ARGS_CAPTURED,
|
|
321
|
+
TOOL_LATENCY_MS,
|
|
316
322
|
TOOL_NAME,
|
|
317
323
|
TOOL_NAME_ATTR_KEYS,
|
|
318
324
|
TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
|
|
319
325
|
TraceFileMissingError,
|
|
320
326
|
TraceNotFoundError,
|
|
327
|
+
applyToolSpanOtlpAttributes,
|
|
321
328
|
asNumber,
|
|
322
329
|
asString,
|
|
323
330
|
buildTraceAnalystTools,
|
|
@@ -330,7 +337,7 @@ import {
|
|
|
330
337
|
stringField,
|
|
331
338
|
traceAnalystFunctionGroup,
|
|
332
339
|
traceSpanKindToOpenInferenceKind
|
|
333
|
-
} from "./chunk-
|
|
340
|
+
} from "./chunk-S2F4J57L.js";
|
|
334
341
|
import {
|
|
335
342
|
RunIntegrityError,
|
|
336
343
|
assertRunCaptured,
|
|
@@ -373,15 +380,39 @@ import {
|
|
|
373
380
|
import {
|
|
374
381
|
LlmCallError,
|
|
375
382
|
LlmClient,
|
|
383
|
+
LlmResponseError,
|
|
376
384
|
LlmRouteAssertionError,
|
|
377
385
|
assertLlmRoute,
|
|
378
386
|
backoffMs,
|
|
379
387
|
callLlm,
|
|
380
388
|
callLlmJson,
|
|
389
|
+
costReceiptFromLlm,
|
|
390
|
+
costReceiptFromLlmError,
|
|
381
391
|
isTransientLlmError,
|
|
392
|
+
maximumChargeForLlmRequest,
|
|
382
393
|
probeLlm,
|
|
383
394
|
stripFencedJson
|
|
384
|
-
} from "./chunk-
|
|
395
|
+
} from "./chunk-NJC7U437.js";
|
|
396
|
+
import {
|
|
397
|
+
CostAccountingIncompleteError,
|
|
398
|
+
CostCallConflictError,
|
|
399
|
+
CostCeilingReachedError,
|
|
400
|
+
CostLedger,
|
|
401
|
+
CostLedgerPersistenceError,
|
|
402
|
+
CostReceiptCaptureError,
|
|
403
|
+
CostReservationExceededError,
|
|
404
|
+
costForUsage,
|
|
405
|
+
modelPriceKey
|
|
406
|
+
} from "./chunk-VCTY3W6J.js";
|
|
407
|
+
import {
|
|
408
|
+
MODEL_PRICING,
|
|
409
|
+
MetricsCollector,
|
|
410
|
+
TokenCounter,
|
|
411
|
+
estimateCost,
|
|
412
|
+
estimateTokens,
|
|
413
|
+
isModelPriced,
|
|
414
|
+
resolveModelPricing
|
|
415
|
+
} from "./chunk-VI2UW6B6.js";
|
|
385
416
|
import {
|
|
386
417
|
FileSystemRawProviderSink,
|
|
387
418
|
InMemoryRawProviderSink,
|
|
@@ -707,6 +738,12 @@ function errMessage(err) {
|
|
|
707
738
|
async function executeScenario(tc, scenario, config) {
|
|
708
739
|
const startTime = Date.now();
|
|
709
740
|
const model = config.model ?? "gpt-4o";
|
|
741
|
+
const costLedger = config.costLedger ?? new CostLedger();
|
|
742
|
+
const costTags = {
|
|
743
|
+
...config.costTags,
|
|
744
|
+
scenarioId: scenario.id,
|
|
745
|
+
executionId: globalThis.crypto.randomUUID()
|
|
746
|
+
};
|
|
710
747
|
const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
|
|
711
748
|
const messages = [{ role: "system", content: systemPrompt }];
|
|
712
749
|
const turns = [];
|
|
@@ -718,12 +755,25 @@ async function executeScenario(tc, scenario, config) {
|
|
|
718
755
|
const turn = scenario.turns[i];
|
|
719
756
|
const turnStart = Date.now();
|
|
720
757
|
messages.push({ role: "user", content: turn.user });
|
|
721
|
-
const
|
|
758
|
+
const request = {
|
|
722
759
|
model,
|
|
723
760
|
messages,
|
|
724
761
|
temperature: 0.4,
|
|
725
762
|
maxTokens: 3e3
|
|
763
|
+
};
|
|
764
|
+
const paid = await costLedger.runPaidCall({
|
|
765
|
+
channel: "agent",
|
|
766
|
+
phase: config.costPhase ?? "benchmark.agent",
|
|
767
|
+
actor: "scenario-agent",
|
|
768
|
+
model,
|
|
769
|
+
maximumCharge: maximumChargeForTCloudRequest(request, config.tcloudMaximumAttempts),
|
|
770
|
+
tags: costTags,
|
|
771
|
+
signal: config.signal,
|
|
772
|
+
execute: () => tc.chat(request),
|
|
773
|
+
receipt: (response) => costReceiptFromTCloud(response, model)
|
|
726
774
|
});
|
|
775
|
+
if (!paid.succeeded) throw paid.error;
|
|
776
|
+
const resp = paid.value;
|
|
727
777
|
const message = resp.choices?.[0]?.message;
|
|
728
778
|
const rawContent = message?.content;
|
|
729
779
|
if (message === void 0 || message === null || typeof rawContent !== "string") {
|
|
@@ -809,7 +859,16 @@ async function executeScenario(tc, scenario, config) {
|
|
|
809
859
|
};
|
|
810
860
|
}
|
|
811
861
|
});
|
|
812
|
-
const judgeInput = {
|
|
862
|
+
const judgeInput = {
|
|
863
|
+
scenario,
|
|
864
|
+
turns,
|
|
865
|
+
artifacts,
|
|
866
|
+
costLedger,
|
|
867
|
+
costPhase: config.costPhase ?? "benchmark.judge",
|
|
868
|
+
costTags,
|
|
869
|
+
signal: config.signal,
|
|
870
|
+
tcloudMaximumAttempts: config.tcloudMaximumAttempts
|
|
871
|
+
};
|
|
813
872
|
const judgeResults = [];
|
|
814
873
|
let failedJudges = 0;
|
|
815
874
|
const judgeFailures = [];
|
|
@@ -878,7 +937,8 @@ async function executeScenario(tc, scenario, config) {
|
|
|
878
937
|
judgeErrors: errorScores.length + failedJudges,
|
|
879
938
|
overallScore,
|
|
880
939
|
totalDurationMs: Date.now() - startTime,
|
|
881
|
-
artifacts
|
|
940
|
+
artifacts,
|
|
941
|
+
cost: costLedger.summary({ tags: costTags })
|
|
882
942
|
};
|
|
883
943
|
if (judgeFailures.length > 0) result.judgeFailures = judgeFailures;
|
|
884
944
|
return result;
|
|
@@ -895,6 +955,8 @@ var BenchmarkRunner = class {
|
|
|
895
955
|
async run(scenarios) {
|
|
896
956
|
const toRun = scenarios ?? this.config.scenarios;
|
|
897
957
|
const passThreshold = this.config.passThreshold ?? 6;
|
|
958
|
+
const costLedger = this.config.costLedger ?? new CostLedger();
|
|
959
|
+
const costTags = { benchmarkRunId: globalThis.crypto.randomUUID() };
|
|
898
960
|
console.log("=".repeat(70));
|
|
899
961
|
console.log(" AGENT EVAL \u2014 BENCHMARK");
|
|
900
962
|
console.log(" Multi-turn scenarios x Multi-judge panel");
|
|
@@ -912,7 +974,10 @@ var BenchmarkRunner = class {
|
|
|
912
974
|
const result = await executeScenario(this.tc, scenario, {
|
|
913
975
|
systemPrompt: this.config.systemPrompt,
|
|
914
976
|
model: this.config.model,
|
|
915
|
-
judges: this.config.judges
|
|
977
|
+
judges: this.config.judges,
|
|
978
|
+
costLedger,
|
|
979
|
+
costTags,
|
|
980
|
+
tcloudMaximumAttempts: this.config.tcloudMaximumAttempts
|
|
916
981
|
});
|
|
917
982
|
results.push(result);
|
|
918
983
|
for (const turn of result.turns) {
|
|
@@ -1012,6 +1077,7 @@ var BenchmarkRunner = class {
|
|
|
1012
1077
|
promptVersion: this.config.promptVersion ?? "v1",
|
|
1013
1078
|
scenarioCount: toRun.length,
|
|
1014
1079
|
results,
|
|
1080
|
+
cost: costLedger.summary({ tags: costTags }),
|
|
1015
1081
|
summary: { overallAvg, byPersona, byDimension, weakest, strongest }
|
|
1016
1082
|
};
|
|
1017
1083
|
}
|
|
@@ -1618,11 +1684,15 @@ var AgentDriver = class {
|
|
|
1618
1684
|
client;
|
|
1619
1685
|
driverModel;
|
|
1620
1686
|
productContext;
|
|
1687
|
+
costLedger;
|
|
1688
|
+
tcloudMaximumAttempts;
|
|
1621
1689
|
constructor(tc, config) {
|
|
1622
1690
|
this.tc = tc;
|
|
1623
1691
|
this.client = config.client;
|
|
1624
1692
|
this.driverModel = config.driverModel ?? "claude-sonnet-4-6";
|
|
1625
1693
|
this.productContext = config.productContext ?? "";
|
|
1694
|
+
this.costLedger = config.costLedger ?? new CostLedger();
|
|
1695
|
+
this.tcloudMaximumAttempts = config.tcloudMaximumAttempts;
|
|
1626
1696
|
}
|
|
1627
1697
|
/**
|
|
1628
1698
|
* Run a persona through the product.
|
|
@@ -1631,6 +1701,7 @@ var AgentDriver = class {
|
|
|
1631
1701
|
* quality curve, and convergence curve.
|
|
1632
1702
|
*/
|
|
1633
1703
|
async run(persona) {
|
|
1704
|
+
const costTags = { driverRunId: globalThis.crypto.randomUUID() };
|
|
1634
1705
|
const email = `eval-driver-${Date.now()}@test.agent-eval.local`;
|
|
1635
1706
|
await this.client.signup(`Driver ${persona.role}`, email, "eval-driver-pass");
|
|
1636
1707
|
await this.client.login(email, "eval-driver-pass");
|
|
@@ -1645,7 +1716,12 @@ var AgentDriver = class {
|
|
|
1645
1716
|
let criteriaMetAtTurn = null;
|
|
1646
1717
|
for (let turn = 1; turn <= persona.maxTurns; turn++) {
|
|
1647
1718
|
const state = await metrics.getState();
|
|
1648
|
-
const userMessage = await this.decideNextMessage(
|
|
1719
|
+
const userMessage = await this.decideNextMessage(
|
|
1720
|
+
persona,
|
|
1721
|
+
state,
|
|
1722
|
+
conversationHistory,
|
|
1723
|
+
costTags
|
|
1724
|
+
);
|
|
1649
1725
|
if (userMessage === "DONE") {
|
|
1650
1726
|
completed = true;
|
|
1651
1727
|
turnsToCompletion = turn - 1;
|
|
@@ -1693,18 +1769,21 @@ var AgentDriver = class {
|
|
|
1693
1769
|
metrics: turnMetrics,
|
|
1694
1770
|
finalState,
|
|
1695
1771
|
convergenceCurve: convergence.getCurve(),
|
|
1696
|
-
totalCostUsd:
|
|
1772
|
+
totalCostUsd: this.costLedger.summary({ tags: costTags }).totalCostUsd,
|
|
1697
1773
|
finalQualityScore: null
|
|
1698
1774
|
};
|
|
1699
1775
|
}
|
|
1700
1776
|
/** Use the driver LLM to decide what the "user" says next */
|
|
1701
|
-
async decideNextMessage(persona, state, history) {
|
|
1777
|
+
async decideNextMessage(persona, state, history, costTags) {
|
|
1702
1778
|
return decideNextUserTurn(this.tc, {
|
|
1703
1779
|
persona,
|
|
1704
1780
|
state,
|
|
1705
1781
|
history,
|
|
1706
1782
|
productContext: this.productContext,
|
|
1707
|
-
model: this.driverModel
|
|
1783
|
+
model: this.driverModel,
|
|
1784
|
+
costLedger: this.costLedger,
|
|
1785
|
+
costTags,
|
|
1786
|
+
tcloudMaximumAttempts: this.tcloudMaximumAttempts
|
|
1708
1787
|
});
|
|
1709
1788
|
}
|
|
1710
1789
|
/** Handle pending approvals based on persona feedback patterns */
|
|
@@ -1813,7 +1892,7 @@ async function decideNextUserTurn(tc, opts) {
|
|
|
1813
1892
|
const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
|
|
1814
1893
|
const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet \u2014 this is the first message)";
|
|
1815
1894
|
const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
|
|
1816
|
-
const
|
|
1895
|
+
const request = {
|
|
1817
1896
|
model,
|
|
1818
1897
|
messages: [
|
|
1819
1898
|
{ role: "system", content: buildDriverSystemPrompt(persona, state, productContext) },
|
|
@@ -1828,7 +1907,19 @@ ${lastResponse}` : "No conversation yet. Send your opening message \u2014 in cha
|
|
|
1828
1907
|
],
|
|
1829
1908
|
temperature: 0.5,
|
|
1830
1909
|
maxTokens: 700
|
|
1910
|
+
};
|
|
1911
|
+
const paid = await (opts.costLedger ?? new CostLedger()).runPaidCall({
|
|
1912
|
+
channel: "driver",
|
|
1913
|
+
phase: "driver-turn",
|
|
1914
|
+
actor: "decideNextUserTurn",
|
|
1915
|
+
model,
|
|
1916
|
+
tags: opts.costTags,
|
|
1917
|
+
maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
|
|
1918
|
+
execute: () => tc.chat(request),
|
|
1919
|
+
receipt: (response) => costReceiptFromTCloud(response, model)
|
|
1831
1920
|
});
|
|
1921
|
+
if (!paid.succeeded) throw paid.error;
|
|
1922
|
+
const resp = paid.value;
|
|
1832
1923
|
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
1833
1924
|
return content.trim();
|
|
1834
1925
|
}
|
|
@@ -3791,7 +3882,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3791
3882
|
const dimAcc = {};
|
|
3792
3883
|
for (const d of dimensionKeys) dimAcc[d] = [];
|
|
3793
3884
|
let rationale = "";
|
|
3794
|
-
let costUsd = 0;
|
|
3795
3885
|
const seenCount = /* @__PURE__ */ new Map();
|
|
3796
3886
|
const keyFor = (model) => {
|
|
3797
3887
|
const n = (seenCount.get(model) ?? 0) + 1;
|
|
@@ -3799,7 +3889,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3799
3889
|
return n === 1 ? model : `${model}#${n}`;
|
|
3800
3890
|
};
|
|
3801
3891
|
for (const v of verdicts) {
|
|
3802
|
-
costUsd += v.costUsd ?? 0;
|
|
3803
3892
|
const key = keyFor(v.model);
|
|
3804
3893
|
if (!v.perDimension) {
|
|
3805
3894
|
failedJudges.push(key);
|
|
@@ -3841,7 +3930,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3841
3930
|
perJudge,
|
|
3842
3931
|
maxDisagreement,
|
|
3843
3932
|
failedJudges,
|
|
3844
|
-
costUsd,
|
|
3845
3933
|
rationale: rationale || "llm-judge",
|
|
3846
3934
|
verdicts: [...verdicts]
|
|
3847
3935
|
};
|
|
@@ -3921,37 +4009,92 @@ function ensembleJudge(opts) {
|
|
|
3921
4009
|
if (opts.crossFamily !== false) {
|
|
3922
4010
|
assertCrossFamily(opts.models);
|
|
3923
4011
|
}
|
|
3924
|
-
const
|
|
3925
|
-
|
|
3926
|
-
|
|
3927
|
-
|
|
3928
|
-
|
|
3929
|
-
|
|
3930
|
-
|
|
3931
|
-
|
|
4012
|
+
const declaredJudgeVersion = opts.judgeVersion?.trim();
|
|
4013
|
+
if (opts.judgeVersion !== void 0 && !declaredJudgeVersion) {
|
|
4014
|
+
throw new Error(`ensembleJudge '${opts.name}': judgeVersion must be non-empty when provided`);
|
|
4015
|
+
}
|
|
4016
|
+
const judgeVersion = declaredJudgeVersion ?? contentHash({
|
|
4017
|
+
kind: "ensembleJudge",
|
|
4018
|
+
models: opts.models,
|
|
4019
|
+
dimensions: opts.dimensions,
|
|
4020
|
+
weights: opts.weights ?? null,
|
|
4021
|
+
crossFamily: opts.crossFamily ?? true,
|
|
4022
|
+
maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge.toString() : opts.maximumCharge ?? null,
|
|
4023
|
+
retry: opts.retry ? {
|
|
4024
|
+
maxAttempts: opts.retry.maxAttempts ?? null,
|
|
4025
|
+
timeoutMs: opts.retry.timeoutMs ?? null,
|
|
4026
|
+
models: opts.retry.models ?? null,
|
|
4027
|
+
backoffMs: opts.retry.backoffMs?.toString() ?? null,
|
|
4028
|
+
isRetryable: opts.retry.isRetryable?.toString() ?? null
|
|
4029
|
+
} : null,
|
|
4030
|
+
scoreWith: opts.scoreWith.toString()
|
|
4031
|
+
});
|
|
4032
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
4033
|
+
const scoreOne = async (args) => {
|
|
4034
|
+
const outcome = await withJudgeRetry(
|
|
4035
|
+
async (model, retrySignal) => {
|
|
4036
|
+
const paid = await args.costLedger.runPaidCall({
|
|
4037
|
+
channel: "judge",
|
|
4038
|
+
phase: args.costPhase,
|
|
4039
|
+
actor: `${opts.name}.${model}`,
|
|
3932
4040
|
model,
|
|
3933
|
-
|
|
3934
|
-
|
|
3935
|
-
|
|
3936
|
-
|
|
3937
|
-
|
|
3938
|
-
|
|
3939
|
-
|
|
3940
|
-
|
|
3941
|
-
|
|
4041
|
+
maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge(model) : opts.maximumCharge,
|
|
4042
|
+
tags: args.costTags,
|
|
4043
|
+
signal: AbortSignal.any([args.signal, retrySignal]),
|
|
4044
|
+
execute: (signal) => opts.scoreWith(model, { artifact: args.artifact, scenario: args.scenario, signal }),
|
|
4045
|
+
receipt: (verdict) => {
|
|
4046
|
+
const cachedTokens = verdict.usage?.cachedPromptTokens ?? 0;
|
|
4047
|
+
const usageUnknown = !verdict.usage || verdict.usage.captured === false;
|
|
4048
|
+
return {
|
|
4049
|
+
model: verdict.model,
|
|
4050
|
+
inputTokens: Math.max(0, (verdict.usage?.promptTokens ?? 0) - cachedTokens),
|
|
4051
|
+
outputTokens: verdict.usage?.completionTokens ?? 0,
|
|
4052
|
+
cachedTokens: cachedTokens > 0 ? cachedTokens : void 0,
|
|
4053
|
+
usageUnknown,
|
|
4054
|
+
...verdict.costUsd === void 0 ? {} : { actualCostUsd: verdict.costUsd }
|
|
4055
|
+
};
|
|
4056
|
+
},
|
|
4057
|
+
receiptFromError: (error) => opts.receiptFromError?.(error, model)
|
|
4058
|
+
});
|
|
4059
|
+
if (!paid.succeeded) throw paid.error;
|
|
4060
|
+
return paid.value;
|
|
4061
|
+
},
|
|
4062
|
+
opts.retry ? { ...opts.retry, models: [args.model] } : { maxAttempts: 1, models: [args.model], isRetryable: () => false }
|
|
4063
|
+
);
|
|
4064
|
+
if (!outcome.succeeded || outcome.value === null) {
|
|
3942
4065
|
return {
|
|
3943
|
-
model,
|
|
4066
|
+
model: args.model,
|
|
3944
4067
|
perDimension: null,
|
|
3945
|
-
rationale:
|
|
4068
|
+
rationale: outcome.error?.message ?? "judge failed"
|
|
3946
4069
|
};
|
|
3947
4070
|
}
|
|
4071
|
+
return outcome.value;
|
|
3948
4072
|
};
|
|
3949
4073
|
return {
|
|
3950
4074
|
name: opts.name,
|
|
3951
4075
|
dimensions: opts.dimensions.map((d) => ({ key: d, description: d })),
|
|
3952
|
-
|
|
3953
|
-
|
|
3954
|
-
|
|
4076
|
+
judgeVersion,
|
|
4077
|
+
async score({
|
|
4078
|
+
artifact,
|
|
4079
|
+
scenario,
|
|
4080
|
+
signal,
|
|
4081
|
+
costLedger,
|
|
4082
|
+
costPhase,
|
|
4083
|
+
costTags
|
|
4084
|
+
}) {
|
|
4085
|
+
const verdicts = await Promise.all(
|
|
4086
|
+
opts.models.map(
|
|
4087
|
+
(model) => scoreOne({
|
|
4088
|
+
model,
|
|
4089
|
+
artifact,
|
|
4090
|
+
scenario,
|
|
4091
|
+
signal,
|
|
4092
|
+
costLedger: costLedger ?? directCostLedger,
|
|
4093
|
+
costPhase: costPhase ?? "judge",
|
|
4094
|
+
costTags
|
|
4095
|
+
})
|
|
4096
|
+
)
|
|
4097
|
+
);
|
|
3955
4098
|
const agg = aggregateJudgeVerdicts(verdicts, opts.dimensions, opts.weights);
|
|
3956
4099
|
const score = {
|
|
3957
4100
|
dimensions: agg.perDimension,
|
|
@@ -4567,118 +4710,13 @@ function clampUnit(value) {
|
|
|
4567
4710
|
return Math.max(0, Math.min(1, value));
|
|
4568
4711
|
}
|
|
4569
4712
|
|
|
4570
|
-
// src/cost-ledger.ts
|
|
4571
|
-
function modelPriceKey(model) {
|
|
4572
|
-
return isModelPriced(model) ? model : null;
|
|
4573
|
-
}
|
|
4574
|
-
function costForUsage(model, usage) {
|
|
4575
|
-
assertNonNegative(usage.inputTokens, "inputTokens");
|
|
4576
|
-
assertNonNegative(usage.outputTokens, "outputTokens");
|
|
4577
|
-
if (usage.cachedTokens !== void 0) assertNonNegative(usage.cachedTokens, "cachedTokens");
|
|
4578
|
-
const pricing = resolveModelPricing(model);
|
|
4579
|
-
if (!pricing) return { costUsd: 0, costUnknown: true };
|
|
4580
|
-
const billedInput = usage.inputTokens + (usage.cachedTokens ?? 0);
|
|
4581
|
-
return { costUsd: estimateCost(billedInput, usage.outputTokens, model), costUnknown: false };
|
|
4582
|
-
}
|
|
4583
|
-
var CostLedger = class {
|
|
4584
|
-
entries = [];
|
|
4585
|
-
completedTasks = 0;
|
|
4586
|
-
/**
|
|
4587
|
-
* Record one LLM call. The cost is computed from pricing unless
|
|
4588
|
-
* `actualCostUsd` is supplied (a finite observed cost from the provider
|
|
4589
|
-
* response), in which case `costUnknown` is false regardless of pricing.
|
|
4590
|
-
*/
|
|
4591
|
-
record(input) {
|
|
4592
|
-
const { costUsd, costUnknown } = costForUsage(input.model, input.usage);
|
|
4593
|
-
const hasActual = typeof input.actualCostUsd === "number" && Number.isFinite(input.actualCostUsd);
|
|
4594
|
-
if (hasActual) assertNonNegative(input.actualCostUsd, "actualCostUsd");
|
|
4595
|
-
const entry = {
|
|
4596
|
-
model: input.model,
|
|
4597
|
-
channel: input.channel,
|
|
4598
|
-
inputTokens: input.usage.inputTokens,
|
|
4599
|
-
outputTokens: input.usage.outputTokens,
|
|
4600
|
-
cachedTokens: input.usage.cachedTokens,
|
|
4601
|
-
costUsd: hasActual ? input.actualCostUsd : costUsd,
|
|
4602
|
-
costUnknown: hasActual ? false : costUnknown,
|
|
4603
|
-
actualCostUsd: hasActual ? input.actualCostUsd : void 0,
|
|
4604
|
-
tags: input.tags,
|
|
4605
|
-
timestamp: input.timestamp ?? Date.now()
|
|
4606
|
-
};
|
|
4607
|
-
this.entries.push(entry);
|
|
4608
|
-
return entry;
|
|
4609
|
-
}
|
|
4610
|
-
/** Increment the completed-task counter (used for cost-per-completed-task). */
|
|
4611
|
-
markCompleted(count = 1) {
|
|
4612
|
-
if (!Number.isInteger(count) || count < 0) {
|
|
4613
|
-
throw new ValidationError(
|
|
4614
|
-
`CostLedger.markCompleted: count must be a non-negative integer, got ${count}`
|
|
4615
|
-
);
|
|
4616
|
-
}
|
|
4617
|
-
this.completedTasks += count;
|
|
4618
|
-
}
|
|
4619
|
-
list() {
|
|
4620
|
-
return [...this.entries];
|
|
4621
|
-
}
|
|
4622
|
-
summary() {
|
|
4623
|
-
const byChannel = /* @__PURE__ */ new Map();
|
|
4624
|
-
const unpriced = /* @__PURE__ */ new Set();
|
|
4625
|
-
let totalCost = 0;
|
|
4626
|
-
let inputTokens = 0;
|
|
4627
|
-
let outputTokens = 0;
|
|
4628
|
-
let cachedTokens = 0;
|
|
4629
|
-
for (const e of this.entries) {
|
|
4630
|
-
totalCost += e.costUsd;
|
|
4631
|
-
inputTokens += e.inputTokens;
|
|
4632
|
-
outputTokens += e.outputTokens;
|
|
4633
|
-
cachedTokens += e.cachedTokens ?? 0;
|
|
4634
|
-
if (e.costUnknown) unpriced.add(e.model);
|
|
4635
|
-
const roll = byChannel.get(e.channel) ?? {
|
|
4636
|
-
channel: e.channel,
|
|
4637
|
-
calls: 0,
|
|
4638
|
-
inputTokens: 0,
|
|
4639
|
-
outputTokens: 0,
|
|
4640
|
-
cachedTokens: 0,
|
|
4641
|
-
costUsd: 0,
|
|
4642
|
-
unpricedCalls: 0
|
|
4643
|
-
};
|
|
4644
|
-
roll.calls += 1;
|
|
4645
|
-
roll.inputTokens += e.inputTokens;
|
|
4646
|
-
roll.outputTokens += e.outputTokens;
|
|
4647
|
-
roll.cachedTokens += e.cachedTokens ?? 0;
|
|
4648
|
-
roll.costUsd += e.costUsd;
|
|
4649
|
-
if (e.costUnknown) roll.unpricedCalls += 1;
|
|
4650
|
-
byChannel.set(e.channel, roll);
|
|
4651
|
-
}
|
|
4652
|
-
return {
|
|
4653
|
-
totalCalls: this.entries.length,
|
|
4654
|
-
inputTokens,
|
|
4655
|
-
outputTokens,
|
|
4656
|
-
cachedTokens,
|
|
4657
|
-
totalCostUsd: totalCost,
|
|
4658
|
-
byChannel: [...byChannel.values()].sort((a, b) => a.channel.localeCompare(b.channel)),
|
|
4659
|
-
unpricedModels: [...unpriced].sort(),
|
|
4660
|
-
fullyPriced: unpriced.size === 0
|
|
4661
|
-
};
|
|
4662
|
-
}
|
|
4663
|
-
/** Total spend divided by completed tasks; null when nothing completed. */
|
|
4664
|
-
costPerCompletedTask() {
|
|
4665
|
-
if (this.completedTasks === 0) return null;
|
|
4666
|
-
return this.summary().totalCostUsd / this.completedTasks;
|
|
4667
|
-
}
|
|
4668
|
-
};
|
|
4669
|
-
function assertNonNegative(n, name) {
|
|
4670
|
-
if (!Number.isFinite(n) || n < 0) {
|
|
4671
|
-
throw new ValidationError(`CostLedger: ${name} must be a non-negative finite number, got ${n}`);
|
|
4672
|
-
}
|
|
4673
|
-
}
|
|
4674
|
-
|
|
4675
4713
|
// src/cost-tracker.ts
|
|
4676
4714
|
var CostTracker = class {
|
|
4677
4715
|
byScenario = /* @__PURE__ */ new Map();
|
|
4678
4716
|
record(entry) {
|
|
4679
4717
|
const full = { timestamp: entry.timestamp ?? Date.now(), ...entry };
|
|
4680
|
-
|
|
4681
|
-
|
|
4718
|
+
assertNonNegative(full.inputTokens, "inputTokens");
|
|
4719
|
+
assertNonNegative(full.outputTokens, "outputTokens");
|
|
4682
4720
|
let bucket = this.byScenario.get(full.scenarioId);
|
|
4683
4721
|
if (!bucket) {
|
|
4684
4722
|
bucket = {
|
|
@@ -4757,7 +4795,7 @@ function costFor(entry) {
|
|
|
4757
4795
|
}
|
|
4758
4796
|
return estimateCost(entry.inputTokens, entry.outputTokens, entry.model);
|
|
4759
4797
|
}
|
|
4760
|
-
function
|
|
4798
|
+
function assertNonNegative(n, name) {
|
|
4761
4799
|
if (!Number.isFinite(n) || n < 0) {
|
|
4762
4800
|
throw new Error(`CostTracker: ${name} must be a non-negative finite number, got ${n}`);
|
|
4763
4801
|
}
|
|
@@ -7983,6 +8021,7 @@ function flowLayer(input) {
|
|
|
7983
8021
|
var INTENT_MATCH_JUDGE_VERSION = "intent-match-judge-v1-2026-04-24";
|
|
7984
8022
|
var DEFAULT_MODEL = "claude-sonnet-4-6";
|
|
7985
8023
|
var DEFAULT_TIMEOUT = 3e5;
|
|
8024
|
+
var DEFAULT_MAX_TOKENS = 800;
|
|
7986
8025
|
var DEFAULT_MAX_SOURCE = 25e3;
|
|
7987
8026
|
var DEFAULT_MAX_PER_FILE = 12e3;
|
|
7988
8027
|
var DEFAULT_MAX_HTML = 2e4;
|
|
@@ -8052,10 +8091,14 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8052
8091
|
const opts = {
|
|
8053
8092
|
model: options.model ?? DEFAULT_MODEL,
|
|
8054
8093
|
timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
|
|
8094
|
+
maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
8055
8095
|
maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
|
|
8056
8096
|
maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
|
|
8057
8097
|
maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
|
|
8058
|
-
llm: options.llm ?? {}
|
|
8098
|
+
llm: options.llm ?? {},
|
|
8099
|
+
costLedger: options.costLedger ?? new CostLedger(),
|
|
8100
|
+
costPhase: options.costPhase ?? "judge.intent-match",
|
|
8101
|
+
signal: options.signal ?? new AbortController().signal
|
|
8059
8102
|
};
|
|
8060
8103
|
if (input.sourceFiles.length === 0 && !input.servedHtml) {
|
|
8061
8104
|
return {
|
|
@@ -8069,23 +8112,40 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8069
8112
|
error: "no input artifact"
|
|
8070
8113
|
};
|
|
8071
8114
|
}
|
|
8115
|
+
let receipt;
|
|
8072
8116
|
try {
|
|
8073
|
-
const
|
|
8074
|
-
|
|
8075
|
-
|
|
8076
|
-
|
|
8077
|
-
|
|
8078
|
-
|
|
8079
|
-
|
|
8080
|
-
|
|
8081
|
-
|
|
8082
|
-
|
|
8083
|
-
|
|
8084
|
-
|
|
8085
|
-
|
|
8086
|
-
|
|
8087
|
-
|
|
8088
|
-
|
|
8117
|
+
const request = {
|
|
8118
|
+
model: opts.model,
|
|
8119
|
+
messages: [
|
|
8120
|
+
{
|
|
8121
|
+
role: "system",
|
|
8122
|
+
content: "You are a holistic code reviewer answering one question: did the agent build the right app for the user. Return strict JSON. No prose outside."
|
|
8123
|
+
},
|
|
8124
|
+
{ role: "user", content: buildPrompt(input, opts) }
|
|
8125
|
+
],
|
|
8126
|
+
jsonSchema: { name: "intent_match_judge", schema: INTENT_SCHEMA },
|
|
8127
|
+
temperature: 0,
|
|
8128
|
+
maxTokens: opts.maxTokens,
|
|
8129
|
+
timeoutMs: opts.timeoutMs
|
|
8130
|
+
};
|
|
8131
|
+
const paid = await opts.costLedger.runPaidCall({
|
|
8132
|
+
channel: "judge",
|
|
8133
|
+
phase: opts.costPhase,
|
|
8134
|
+
actor: "intent-match",
|
|
8135
|
+
model: opts.model,
|
|
8136
|
+
maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
|
|
8137
|
+
signal: opts.signal,
|
|
8138
|
+
execute: (signal, callId) => callLlmJson(request, {
|
|
8139
|
+
...opts.llm,
|
|
8140
|
+
signal,
|
|
8141
|
+
idempotencyKey: callId
|
|
8142
|
+
}),
|
|
8143
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
8144
|
+
receiptFromError: costReceiptFromLlmError
|
|
8145
|
+
});
|
|
8146
|
+
receipt = paid.receipt;
|
|
8147
|
+
if (!paid.succeeded) throw paid.error;
|
|
8148
|
+
const { value } = paid.value;
|
|
8089
8149
|
const score = Math.max(0, Math.min(1, Number(value?.score ?? 0)));
|
|
8090
8150
|
return {
|
|
8091
8151
|
kind: "intent-match",
|
|
@@ -8093,7 +8153,7 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8093
8153
|
score: Number(score.toFixed(3)),
|
|
8094
8154
|
evidence: String(value?.evidence ?? "").slice(0, 400),
|
|
8095
8155
|
durationMs: Date.now() - start,
|
|
8096
|
-
costUsd:
|
|
8156
|
+
costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
|
|
8097
8157
|
available: true
|
|
8098
8158
|
};
|
|
8099
8159
|
} catch (err) {
|
|
@@ -8103,7 +8163,7 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8103
8163
|
score: 0,
|
|
8104
8164
|
evidence: "",
|
|
8105
8165
|
durationMs: Date.now() - start,
|
|
8106
|
-
costUsd: null,
|
|
8166
|
+
costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
|
|
8107
8167
|
available: false,
|
|
8108
8168
|
error: err instanceof Error ? err.message : String(err)
|
|
8109
8169
|
};
|
|
@@ -9143,23 +9203,42 @@ function createDefaultReviewer(options) {
|
|
|
9143
9203
|
};
|
|
9144
9204
|
const promptBuilder = options.promptBuilder ?? buildReviewerPrompt;
|
|
9145
9205
|
const timeoutMs = options.timeoutMs ?? 3e5;
|
|
9206
|
+
const maxTokens = options.maxTokens ?? 4e3;
|
|
9207
|
+
const costLedger = options.costLedger ?? new CostLedger();
|
|
9146
9208
|
return async (input) => {
|
|
9147
9209
|
const start = Date.now();
|
|
9148
9210
|
const { system, user } = promptBuilder(input);
|
|
9211
|
+
let receipt;
|
|
9149
9212
|
try {
|
|
9150
|
-
const
|
|
9151
|
-
|
|
9152
|
-
|
|
9153
|
-
|
|
9154
|
-
|
|
9155
|
-
|
|
9156
|
-
|
|
9157
|
-
|
|
9158
|
-
|
|
9159
|
-
|
|
9160
|
-
|
|
9161
|
-
|
|
9162
|
-
|
|
9213
|
+
const request = {
|
|
9214
|
+
model: options.model,
|
|
9215
|
+
messages: [
|
|
9216
|
+
{ role: "system", content: system },
|
|
9217
|
+
{ role: "user", content: user }
|
|
9218
|
+
],
|
|
9219
|
+
jsonSchema: { name: "reviewer_output", schema: REVIEWER_SCHEMA },
|
|
9220
|
+
temperature: 0,
|
|
9221
|
+
maxTokens,
|
|
9222
|
+
timeoutMs
|
|
9223
|
+
};
|
|
9224
|
+
const paid = await costLedger.runPaidCall({
|
|
9225
|
+
channel: "analyst",
|
|
9226
|
+
phase: options.costPhase ?? "review",
|
|
9227
|
+
actor: "default-reviewer",
|
|
9228
|
+
model: options.model,
|
|
9229
|
+
maximumCharge: maximumChargeForLlmRequest(request, options.llm),
|
|
9230
|
+
signal: options.signal,
|
|
9231
|
+
execute: (signal, callId) => callLlmJson(request, {
|
|
9232
|
+
...options.llm,
|
|
9233
|
+
signal,
|
|
9234
|
+
idempotencyKey: callId
|
|
9235
|
+
}),
|
|
9236
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
9237
|
+
receiptFromError: costReceiptFromLlmError
|
|
9238
|
+
});
|
|
9239
|
+
receipt = paid.receipt;
|
|
9240
|
+
if (!paid.succeeded) throw paid.error;
|
|
9241
|
+
const { value } = paid.value;
|
|
9163
9242
|
return {
|
|
9164
9243
|
shot: input.shot,
|
|
9165
9244
|
observations: String(value.observations ?? softFail2.observations),
|
|
@@ -9167,7 +9246,7 @@ function createDefaultReviewer(options) {
|
|
|
9167
9246
|
nextShotInstruction: String(value.nextShotInstruction ?? softFail2.nextShotInstruction),
|
|
9168
9247
|
shouldContinue: Boolean(value.shouldContinue),
|
|
9169
9248
|
confidence: Math.max(0, Math.min(1, Number(value.confidence ?? softFail2.confidence))),
|
|
9170
|
-
costUsd:
|
|
9249
|
+
costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
|
|
9171
9250
|
durationMs: Date.now() - start,
|
|
9172
9251
|
available: true
|
|
9173
9252
|
};
|
|
@@ -9179,7 +9258,7 @@ function createDefaultReviewer(options) {
|
|
|
9179
9258
|
nextShotInstruction: softFail2.nextShotInstruction,
|
|
9180
9259
|
shouldContinue: softFail2.shouldContinue,
|
|
9181
9260
|
confidence: softFail2.confidence,
|
|
9182
|
-
costUsd: null,
|
|
9261
|
+
costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
|
|
9183
9262
|
durationMs: Date.now() - start,
|
|
9184
9263
|
available: false,
|
|
9185
9264
|
error: err instanceof Error ? err.message : String(err)
|
|
@@ -10259,18 +10338,23 @@ async function runDistillation(opts) {
|
|
|
10259
10338
|
input: scenario.input,
|
|
10260
10339
|
scenarioId: scenario.id
|
|
10261
10340
|
});
|
|
10262
|
-
const
|
|
10263
|
-
|
|
10264
|
-
|
|
10265
|
-
|
|
10266
|
-
|
|
10267
|
-
|
|
10268
|
-
|
|
10269
|
-
|
|
10270
|
-
|
|
10271
|
-
|
|
10272
|
-
|
|
10273
|
-
|
|
10341
|
+
const request = {
|
|
10342
|
+
model: opts.studentModel,
|
|
10343
|
+
messages: prompt,
|
|
10344
|
+
jsonMode: true,
|
|
10345
|
+
temperature: studentTemperature,
|
|
10346
|
+
maxTokens: studentMaxTokens
|
|
10347
|
+
};
|
|
10348
|
+
const paid = await ctx.cost.runPaidCall({
|
|
10349
|
+
actor: "distillation-student",
|
|
10350
|
+
model: opts.studentModel,
|
|
10351
|
+
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maxRetries: chat.maximumAttempts }),
|
|
10352
|
+
execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
|
|
10353
|
+
receipt: costReceiptFromLlm,
|
|
10354
|
+
receiptFromError: costReceiptFromLlmError
|
|
10355
|
+
});
|
|
10356
|
+
if (!paid.succeeded) throw paid.error;
|
|
10357
|
+
return parse(paid.value.content, scenario.id);
|
|
10274
10358
|
}
|
|
10275
10359
|
});
|
|
10276
10360
|
const winnerPrompt = typeof loop.winnerSurface === "string" ? loop.winnerSurface : opts.baselinePrompt;
|
|
@@ -10282,14 +10366,6 @@ async function runDistillation(opts) {
|
|
|
10282
10366
|
holdoutAgreement: { baseline, winner, delta: winner - baseline }
|
|
10283
10367
|
};
|
|
10284
10368
|
}
|
|
10285
|
-
function reportUsage(cost, response) {
|
|
10286
|
-
if (typeof response.costUsd === "number") cost.observe(response.costUsd, "distillation-student");
|
|
10287
|
-
cost.observeTokens({
|
|
10288
|
-
input: response.usage.promptTokens,
|
|
10289
|
-
output: response.usage.completionTokens,
|
|
10290
|
-
cached: response.usage.cachedPromptTokens
|
|
10291
|
-
});
|
|
10292
|
-
}
|
|
10293
10369
|
var DEFAULT_MUTATION_PRIMITIVES2 = [
|
|
10294
10370
|
"Add an explicit output-schema instruction so the model emits exactly the gold label fields as JSON.",
|
|
10295
10371
|
"Add a one-line decision rule for each verdict field the student keeps getting wrong.",
|
|
@@ -11452,7 +11528,13 @@ export {
|
|
|
11452
11528
|
CaptureIntegrityError,
|
|
11453
11529
|
ConfigError,
|
|
11454
11530
|
ConvergenceTracker,
|
|
11531
|
+
CostAccountingIncompleteError,
|
|
11532
|
+
CostCallConflictError,
|
|
11533
|
+
CostCeilingReachedError,
|
|
11455
11534
|
CostLedger,
|
|
11535
|
+
CostLedgerPersistenceError,
|
|
11536
|
+
CostReceiptCaptureError,
|
|
11537
|
+
CostReservationExceededError,
|
|
11456
11538
|
CostTracker,
|
|
11457
11539
|
CrossFamilyError,
|
|
11458
11540
|
DEFAULT_AGENT_SLOS,
|
|
@@ -11487,6 +11569,7 @@ export {
|
|
|
11487
11569
|
HoldoutAuditor,
|
|
11488
11570
|
HoldoutLockedError,
|
|
11489
11571
|
IMPROVEMENT_KIND_SPEC,
|
|
11572
|
+
INPUT_VALUE,
|
|
11490
11573
|
INTENT_MATCH_JUDGE_VERSION,
|
|
11491
11574
|
InMemoryFeedbackTrajectoryStore,
|
|
11492
11575
|
InMemoryRawProviderSink,
|
|
@@ -11509,6 +11592,7 @@ export {
|
|
|
11509
11592
|
LLM_OUTPUT_TOKEN_ATTR_KEYS,
|
|
11510
11593
|
LlmCallError,
|
|
11511
11594
|
LlmClient,
|
|
11595
|
+
LlmResponseError,
|
|
11512
11596
|
LlmRouteAssertionError,
|
|
11513
11597
|
LockedJsonlAppender,
|
|
11514
11598
|
MODEL_PRICING,
|
|
@@ -11521,14 +11605,18 @@ export {
|
|
|
11521
11605
|
NotFoundError,
|
|
11522
11606
|
OPENINFERENCE_SPAN_KIND,
|
|
11523
11607
|
OTEL_AGENT_EVAL_SCOPE,
|
|
11608
|
+
OUTPUT_VALUE,
|
|
11524
11609
|
OtlpFileTraceStore,
|
|
11525
11610
|
POLICY_EDIT_AXES,
|
|
11611
|
+
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
11526
11612
|
POLICY_EDIT_TARGET_SURFACES,
|
|
11527
11613
|
PairwiseSteeringOptimizer,
|
|
11528
11614
|
PolicyEditValidationError,
|
|
11529
11615
|
ProductClient,
|
|
11530
11616
|
PromptRegistry,
|
|
11531
11617
|
REDACTION_VERSION,
|
|
11618
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
11619
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
11532
11620
|
RESEARCH_REPORT_HARD_PAIR_FLOOR,
|
|
11533
11621
|
ReplayCache,
|
|
11534
11622
|
ReplayCacheMissError,
|
|
@@ -11546,6 +11634,8 @@ export {
|
|
|
11546
11634
|
SkillUsageAnalyst,
|
|
11547
11635
|
SpanNotFoundError,
|
|
11548
11636
|
SubprocessSandboxDriver,
|
|
11637
|
+
TOOL_ARGS_CAPTURED,
|
|
11638
|
+
TOOL_LATENCY_MS,
|
|
11549
11639
|
TOOL_NAME,
|
|
11550
11640
|
TOOL_NAME_ATTR_KEYS,
|
|
11551
11641
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
@@ -11582,6 +11672,7 @@ export {
|
|
|
11582
11672
|
analyzeTraces,
|
|
11583
11673
|
appendScorecard,
|
|
11584
11674
|
applyPolicyEditToSurface,
|
|
11675
|
+
applyToolSpanOtlpAttributes,
|
|
11585
11676
|
argHash,
|
|
11586
11677
|
asNumber,
|
|
11587
11678
|
asString,
|
|
@@ -11674,6 +11765,8 @@ export {
|
|
|
11674
11765
|
corpusInterRaterAgreement,
|
|
11675
11766
|
corpusInterRaterAgreementFromJudgeScores,
|
|
11676
11767
|
costForUsage,
|
|
11768
|
+
costReceiptFromLlm,
|
|
11769
|
+
costReceiptFromLlmError,
|
|
11677
11770
|
costReport,
|
|
11678
11771
|
createAnalystAi,
|
|
11679
11772
|
createAntiSlopJudge,
|
|
@@ -11687,6 +11780,7 @@ export {
|
|
|
11687
11780
|
createLlmReviewer,
|
|
11688
11781
|
createOtelExporter,
|
|
11689
11782
|
createOtelTracingStore,
|
|
11783
|
+
createReferenceEquivalenceJudge,
|
|
11690
11784
|
createReplayFetch,
|
|
11691
11785
|
createSandboxPool,
|
|
11692
11786
|
createSemanticConceptJudge,
|
|
@@ -11777,6 +11871,7 @@ export {
|
|
|
11777
11871
|
groupBy,
|
|
11778
11872
|
groupRunsByAgentProfileCell,
|
|
11779
11873
|
harnessAxisOf,
|
|
11874
|
+
hasCapturedToolArgs,
|
|
11780
11875
|
hashContent,
|
|
11781
11876
|
hashJson,
|
|
11782
11877
|
hashScenarios,
|
|
@@ -11832,9 +11927,11 @@ export {
|
|
|
11832
11927
|
makeEvalTools,
|
|
11833
11928
|
makeFinding,
|
|
11834
11929
|
makePolicyEdit,
|
|
11930
|
+
makePolicyEditCandidateRecord,
|
|
11835
11931
|
mannWhitneyU,
|
|
11836
11932
|
matchGoldens,
|
|
11837
11933
|
matchSpan,
|
|
11934
|
+
maximumChargeForLlmRequest,
|
|
11838
11935
|
mcnemar,
|
|
11839
11936
|
mcnemarPower,
|
|
11840
11937
|
mcnemarRequiredN,
|
|
@@ -11950,6 +12047,7 @@ export {
|
|
|
11950
12047
|
runProposeReview,
|
|
11951
12048
|
runProposeReviewAsControlLoop,
|
|
11952
12049
|
runRecordToProductBenchmarkRecord,
|
|
12050
|
+
runReferenceEquivalenceJudge,
|
|
11953
12051
|
runReferenceReplay,
|
|
11954
12052
|
runScore,
|
|
11955
12053
|
runSelfPlay,
|
|
@@ -12009,6 +12107,7 @@ export {
|
|
|
12009
12107
|
userQuestionsForKnowledgeGaps,
|
|
12010
12108
|
validateAgentProfileCell,
|
|
12011
12109
|
validatePolicyEdit,
|
|
12110
|
+
validatePolicyEditCandidateRecord,
|
|
12012
12111
|
validateProductBenchmarkManifest,
|
|
12013
12112
|
validateProductBenchmarkRecord,
|
|
12014
12113
|
validateProductBenchmarkRun,
|