@tangle-network/agent-eval 0.116.0 → 0.117.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +38 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +54 -30
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-GSW3OBHK.js → chunk-JSJZ4PJ6.js} +406 -726
- package/dist/chunk-JSJZ4PJ6.js.map +1 -0
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -9,12 +9,12 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-FQNLDL4D.js";
|
|
13
13
|
import {
|
|
14
14
|
BENCHMARK_SPLIT_SEED,
|
|
15
15
|
benchmarks_exports,
|
|
16
16
|
deterministicSplit
|
|
17
|
-
} from "./chunk-
|
|
17
|
+
} from "./chunk-JSDVRFAP.js";
|
|
18
18
|
import {
|
|
19
19
|
DEFAULT_RULES,
|
|
20
20
|
buildTrajectory,
|
|
@@ -23,7 +23,7 @@ import {
|
|
|
23
23
|
computeToolUseMetrics,
|
|
24
24
|
iqr,
|
|
25
25
|
welchsTTest
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-ODVOOEWQ.js";
|
|
27
27
|
import {
|
|
28
28
|
analyzeSeries
|
|
29
29
|
} from "./chunk-BOD4O7OF.js";
|
|
@@ -40,40 +40,45 @@ import {
|
|
|
40
40
|
import {
|
|
41
41
|
CODING_HARNESSES,
|
|
42
42
|
HARNESS_NATIVE_MODEL,
|
|
43
|
-
JudgeParseError,
|
|
44
|
-
adversarialJudge,
|
|
45
43
|
agentProfileHash,
|
|
46
44
|
agentProfileId,
|
|
47
45
|
agentProfileModelId,
|
|
48
|
-
codeExecutionJudge,
|
|
49
|
-
coherenceJudge,
|
|
50
46
|
comparePairedArms,
|
|
51
47
|
completionVerdict,
|
|
52
|
-
createCustomJudge,
|
|
53
|
-
createDomainExpertJudge,
|
|
54
48
|
createLlmCorrectnessChecker,
|
|
55
49
|
createTokenRecallChecker,
|
|
56
|
-
defaultJudges,
|
|
57
50
|
expandProfileAxes,
|
|
58
51
|
extractProducedState,
|
|
59
52
|
harnessAxisOf,
|
|
60
|
-
llmJudge,
|
|
61
53
|
pairArms,
|
|
62
54
|
parseCorrectnessResponse,
|
|
63
55
|
verifyCompletion
|
|
64
|
-
} from "./chunk-
|
|
56
|
+
} from "./chunk-JSJZ4PJ6.js";
|
|
65
57
|
import {
|
|
66
58
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
67
59
|
DEFAULT_RED_TEAM_CORPUS,
|
|
68
60
|
Dataset,
|
|
69
61
|
HoldoutLockedError,
|
|
62
|
+
JudgeParseError,
|
|
63
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
64
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
65
|
+
adversarialJudge,
|
|
70
66
|
buildReflectionPrompt,
|
|
71
67
|
campaignMeanComposite,
|
|
68
|
+
codeExecutionJudge,
|
|
69
|
+
coherenceJudge,
|
|
70
|
+
costReceiptFromTCloud,
|
|
71
|
+
createCustomJudge,
|
|
72
|
+
createDomainExpertJudge,
|
|
73
|
+
createReferenceEquivalenceJudge,
|
|
72
74
|
crowdingDistance,
|
|
75
|
+
defaultJudges,
|
|
73
76
|
dominates,
|
|
74
77
|
gepaProposer,
|
|
75
78
|
hashScenarios,
|
|
76
79
|
heldOutGate,
|
|
80
|
+
llmJudge,
|
|
81
|
+
maximumChargeForTCloudRequest,
|
|
77
82
|
paretoFrontier,
|
|
78
83
|
paretoFrontierWithCrowding,
|
|
79
84
|
parseReflectionResponse,
|
|
@@ -81,20 +86,12 @@ import {
|
|
|
81
86
|
redTeamReport,
|
|
82
87
|
runCanaries,
|
|
83
88
|
runImprovementLoop,
|
|
89
|
+
runReferenceEquivalenceJudge,
|
|
84
90
|
scalarScore,
|
|
85
91
|
scoreRedTeamOutput,
|
|
86
92
|
surfaceContentHash,
|
|
87
93
|
toolNamesForRun
|
|
88
|
-
} from "./chunk-
|
|
89
|
-
import {
|
|
90
|
-
MODEL_PRICING,
|
|
91
|
-
MetricsCollector,
|
|
92
|
-
TokenCounter,
|
|
93
|
-
estimateCost,
|
|
94
|
-
estimateTokens,
|
|
95
|
-
isModelPriced,
|
|
96
|
-
resolveModelPricing
|
|
97
|
-
} from "./chunk-VI2UW6B6.js";
|
|
94
|
+
} from "./chunk-HQPHZGL6.js";
|
|
98
95
|
import {
|
|
99
96
|
BackendIntegrityError,
|
|
100
97
|
assertRealBackend,
|
|
@@ -104,7 +101,7 @@ import {
|
|
|
104
101
|
fileVerdictCache,
|
|
105
102
|
inMemoryVerdictCache,
|
|
106
103
|
summarizeBackendIntegrity
|
|
107
|
-
} from "./chunk-
|
|
104
|
+
} from "./chunk-IDZTTFRR.js";
|
|
108
105
|
import {
|
|
109
106
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
110
107
|
FindingsStore,
|
|
@@ -114,24 +111,23 @@ import {
|
|
|
114
111
|
SKILL_USAGE_ANALYST,
|
|
115
112
|
SkillUsageAnalyst,
|
|
116
113
|
createAnalystAi,
|
|
117
|
-
createChatClient,
|
|
118
114
|
createSemanticConceptJudge,
|
|
119
115
|
defaultIsMaterial,
|
|
120
116
|
diffFindings,
|
|
121
117
|
runSemanticConceptJudge
|
|
122
|
-
} from "./chunk-
|
|
118
|
+
} from "./chunk-CCZIVI3F.js";
|
|
123
119
|
import {
|
|
124
120
|
buildDefaultAnalystRegistry,
|
|
125
|
-
computeTraceMetrics
|
|
126
|
-
|
|
121
|
+
computeTraceMetrics,
|
|
122
|
+
createChatClient
|
|
123
|
+
} from "./chunk-VF3XSYTI.js";
|
|
124
|
+
import "./chunk-HHWE3POT.js";
|
|
127
125
|
import {
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
aggregateRunScore,
|
|
131
|
-
clamp01
|
|
132
|
-
} from "./chunk-MPHTT5HE.js";
|
|
126
|
+
Mutex
|
|
127
|
+
} from "./chunk-3YYRZDON.js";
|
|
133
128
|
import {
|
|
134
129
|
AnalystRegistry,
|
|
130
|
+
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
135
131
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
136
132
|
FAILURE_MODE_KIND_SPEC,
|
|
137
133
|
IMPROVEMENT_KIND_SPEC,
|
|
@@ -142,7 +138,9 @@ import {
|
|
|
142
138
|
POLICY_EDIT_TARGET_SURFACES,
|
|
143
139
|
PolicyEditValidationError,
|
|
144
140
|
admitPolicyEdit,
|
|
141
|
+
aggregateRunScore,
|
|
145
142
|
applyPolicyEditToSurface,
|
|
143
|
+
clamp01,
|
|
146
144
|
computeFindingId,
|
|
147
145
|
computePolicyEditId,
|
|
148
146
|
createTraceAnalystKind,
|
|
@@ -156,7 +154,7 @@ import {
|
|
|
156
154
|
scorePolicyEditReadiness,
|
|
157
155
|
validatePolicyEdit,
|
|
158
156
|
validatePolicyEditCandidateRecord
|
|
159
|
-
} from "./chunk-
|
|
157
|
+
} from "./chunk-MGEHEHSN.js";
|
|
160
158
|
import {
|
|
161
159
|
allCriticalPassed,
|
|
162
160
|
controlFailureClassFromVerification,
|
|
@@ -187,7 +185,7 @@ import {
|
|
|
187
185
|
} from "./chunk-MOXWMGPC.js";
|
|
188
186
|
import {
|
|
189
187
|
runEvalCampaign
|
|
190
|
-
} from "./chunk-
|
|
188
|
+
} from "./chunk-GQCZRZ7L.js";
|
|
191
189
|
import "./chunk-ARU2PZFM.js";
|
|
192
190
|
import {
|
|
193
191
|
evaluateInterimReleaseConfidence,
|
|
@@ -270,13 +268,14 @@ import {
|
|
|
270
268
|
scoreTraceInsightReadiness,
|
|
271
269
|
tokenizeDomainWords,
|
|
272
270
|
traceAnalystOnRunComplete
|
|
273
|
-
} from "./chunk-
|
|
271
|
+
} from "./chunk-YZPO4UHR.js";
|
|
274
272
|
import {
|
|
275
273
|
FAILURE_CLASSES,
|
|
276
274
|
TRACE_SCHEMA_VERSION,
|
|
277
275
|
aggregateLlm,
|
|
278
276
|
argHash,
|
|
279
277
|
groupBy,
|
|
278
|
+
hasCapturedToolArgs,
|
|
280
279
|
isJudgeSpan,
|
|
281
280
|
isLlmSpan,
|
|
282
281
|
isRetrievalSpan,
|
|
@@ -287,13 +286,13 @@ import {
|
|
|
287
286
|
runFailureClass,
|
|
288
287
|
runsForScenario,
|
|
289
288
|
toolSpans
|
|
290
|
-
} from "./chunk-
|
|
289
|
+
} from "./chunk-LQUTGLOZ.js";
|
|
291
290
|
import {
|
|
292
291
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
293
292
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
294
293
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
295
294
|
analyzeTraces
|
|
296
|
-
} from "./chunk-
|
|
295
|
+
} from "./chunk-4JLWXDYA.js";
|
|
297
296
|
import {
|
|
298
297
|
DEFAULT_REDACTION_RULES,
|
|
299
298
|
REDACTION_VERSION,
|
|
@@ -302,6 +301,7 @@ import {
|
|
|
302
301
|
} from "./chunk-GGE4NNQT.js";
|
|
303
302
|
import {
|
|
304
303
|
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
304
|
+
INPUT_VALUE,
|
|
305
305
|
LLM_CACHED_TOKENS,
|
|
306
306
|
LLM_CACHED_TOKEN_ATTR_KEYS,
|
|
307
307
|
LLM_COST_ATTR_KEYS,
|
|
@@ -313,14 +313,18 @@ import {
|
|
|
313
313
|
LLM_OUTPUT_TOKENS,
|
|
314
314
|
LLM_OUTPUT_TOKEN_ATTR_KEYS,
|
|
315
315
|
OPENINFERENCE_SPAN_KIND,
|
|
316
|
+
OUTPUT_VALUE,
|
|
316
317
|
OtlpFileTraceStore,
|
|
317
318
|
SPAN_KIND_ATTR_KEYS,
|
|
318
319
|
SpanNotFoundError,
|
|
320
|
+
TOOL_ARGS_CAPTURED,
|
|
321
|
+
TOOL_LATENCY_MS,
|
|
319
322
|
TOOL_NAME,
|
|
320
323
|
TOOL_NAME_ATTR_KEYS,
|
|
321
324
|
TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
|
|
322
325
|
TraceFileMissingError,
|
|
323
326
|
TraceNotFoundError,
|
|
327
|
+
applyToolSpanOtlpAttributes,
|
|
324
328
|
asNumber,
|
|
325
329
|
asString,
|
|
326
330
|
buildTraceAnalystTools,
|
|
@@ -333,7 +337,7 @@ import {
|
|
|
333
337
|
stringField,
|
|
334
338
|
traceAnalystFunctionGroup,
|
|
335
339
|
traceSpanKindToOpenInferenceKind
|
|
336
|
-
} from "./chunk-
|
|
340
|
+
} from "./chunk-S2F4J57L.js";
|
|
337
341
|
import {
|
|
338
342
|
RunIntegrityError,
|
|
339
343
|
assertRunCaptured,
|
|
@@ -376,15 +380,39 @@ import {
|
|
|
376
380
|
import {
|
|
377
381
|
LlmCallError,
|
|
378
382
|
LlmClient,
|
|
383
|
+
LlmResponseError,
|
|
379
384
|
LlmRouteAssertionError,
|
|
380
385
|
assertLlmRoute,
|
|
381
386
|
backoffMs,
|
|
382
387
|
callLlm,
|
|
383
388
|
callLlmJson,
|
|
389
|
+
costReceiptFromLlm,
|
|
390
|
+
costReceiptFromLlmError,
|
|
384
391
|
isTransientLlmError,
|
|
392
|
+
maximumChargeForLlmRequest,
|
|
385
393
|
probeLlm,
|
|
386
394
|
stripFencedJson
|
|
387
|
-
} from "./chunk-
|
|
395
|
+
} from "./chunk-NJC7U437.js";
|
|
396
|
+
import {
|
|
397
|
+
CostAccountingIncompleteError,
|
|
398
|
+
CostCallConflictError,
|
|
399
|
+
CostCeilingReachedError,
|
|
400
|
+
CostLedger,
|
|
401
|
+
CostLedgerPersistenceError,
|
|
402
|
+
CostReceiptCaptureError,
|
|
403
|
+
CostReservationExceededError,
|
|
404
|
+
costForUsage,
|
|
405
|
+
modelPriceKey
|
|
406
|
+
} from "./chunk-VCTY3W6J.js";
|
|
407
|
+
import {
|
|
408
|
+
MODEL_PRICING,
|
|
409
|
+
MetricsCollector,
|
|
410
|
+
TokenCounter,
|
|
411
|
+
estimateCost,
|
|
412
|
+
estimateTokens,
|
|
413
|
+
isModelPriced,
|
|
414
|
+
resolveModelPricing
|
|
415
|
+
} from "./chunk-VI2UW6B6.js";
|
|
388
416
|
import {
|
|
389
417
|
FileSystemRawProviderSink,
|
|
390
418
|
InMemoryRawProviderSink,
|
|
@@ -710,6 +738,12 @@ function errMessage(err) {
|
|
|
710
738
|
async function executeScenario(tc, scenario, config) {
|
|
711
739
|
const startTime = Date.now();
|
|
712
740
|
const model = config.model ?? "gpt-4o";
|
|
741
|
+
const costLedger = config.costLedger ?? new CostLedger();
|
|
742
|
+
const costTags = {
|
|
743
|
+
...config.costTags,
|
|
744
|
+
scenarioId: scenario.id,
|
|
745
|
+
executionId: globalThis.crypto.randomUUID()
|
|
746
|
+
};
|
|
713
747
|
const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
|
|
714
748
|
const messages = [{ role: "system", content: systemPrompt }];
|
|
715
749
|
const turns = [];
|
|
@@ -721,12 +755,25 @@ async function executeScenario(tc, scenario, config) {
|
|
|
721
755
|
const turn = scenario.turns[i];
|
|
722
756
|
const turnStart = Date.now();
|
|
723
757
|
messages.push({ role: "user", content: turn.user });
|
|
724
|
-
const
|
|
758
|
+
const request = {
|
|
725
759
|
model,
|
|
726
760
|
messages,
|
|
727
761
|
temperature: 0.4,
|
|
728
762
|
maxTokens: 3e3
|
|
763
|
+
};
|
|
764
|
+
const paid = await costLedger.runPaidCall({
|
|
765
|
+
channel: "agent",
|
|
766
|
+
phase: config.costPhase ?? "benchmark.agent",
|
|
767
|
+
actor: "scenario-agent",
|
|
768
|
+
model,
|
|
769
|
+
maximumCharge: maximumChargeForTCloudRequest(request, config.tcloudMaximumAttempts),
|
|
770
|
+
tags: costTags,
|
|
771
|
+
signal: config.signal,
|
|
772
|
+
execute: () => tc.chat(request),
|
|
773
|
+
receipt: (response) => costReceiptFromTCloud(response, model)
|
|
729
774
|
});
|
|
775
|
+
if (!paid.succeeded) throw paid.error;
|
|
776
|
+
const resp = paid.value;
|
|
730
777
|
const message = resp.choices?.[0]?.message;
|
|
731
778
|
const rawContent = message?.content;
|
|
732
779
|
if (message === void 0 || message === null || typeof rawContent !== "string") {
|
|
@@ -812,7 +859,16 @@ async function executeScenario(tc, scenario, config) {
|
|
|
812
859
|
};
|
|
813
860
|
}
|
|
814
861
|
});
|
|
815
|
-
const judgeInput = {
|
|
862
|
+
const judgeInput = {
|
|
863
|
+
scenario,
|
|
864
|
+
turns,
|
|
865
|
+
artifacts,
|
|
866
|
+
costLedger,
|
|
867
|
+
costPhase: config.costPhase ?? "benchmark.judge",
|
|
868
|
+
costTags,
|
|
869
|
+
signal: config.signal,
|
|
870
|
+
tcloudMaximumAttempts: config.tcloudMaximumAttempts
|
|
871
|
+
};
|
|
816
872
|
const judgeResults = [];
|
|
817
873
|
let failedJudges = 0;
|
|
818
874
|
const judgeFailures = [];
|
|
@@ -881,7 +937,8 @@ async function executeScenario(tc, scenario, config) {
|
|
|
881
937
|
judgeErrors: errorScores.length + failedJudges,
|
|
882
938
|
overallScore,
|
|
883
939
|
totalDurationMs: Date.now() - startTime,
|
|
884
|
-
artifacts
|
|
940
|
+
artifacts,
|
|
941
|
+
cost: costLedger.summary({ tags: costTags })
|
|
885
942
|
};
|
|
886
943
|
if (judgeFailures.length > 0) result.judgeFailures = judgeFailures;
|
|
887
944
|
return result;
|
|
@@ -898,6 +955,8 @@ var BenchmarkRunner = class {
|
|
|
898
955
|
async run(scenarios) {
|
|
899
956
|
const toRun = scenarios ?? this.config.scenarios;
|
|
900
957
|
const passThreshold = this.config.passThreshold ?? 6;
|
|
958
|
+
const costLedger = this.config.costLedger ?? new CostLedger();
|
|
959
|
+
const costTags = { benchmarkRunId: globalThis.crypto.randomUUID() };
|
|
901
960
|
console.log("=".repeat(70));
|
|
902
961
|
console.log(" AGENT EVAL \u2014 BENCHMARK");
|
|
903
962
|
console.log(" Multi-turn scenarios x Multi-judge panel");
|
|
@@ -915,7 +974,10 @@ var BenchmarkRunner = class {
|
|
|
915
974
|
const result = await executeScenario(this.tc, scenario, {
|
|
916
975
|
systemPrompt: this.config.systemPrompt,
|
|
917
976
|
model: this.config.model,
|
|
918
|
-
judges: this.config.judges
|
|
977
|
+
judges: this.config.judges,
|
|
978
|
+
costLedger,
|
|
979
|
+
costTags,
|
|
980
|
+
tcloudMaximumAttempts: this.config.tcloudMaximumAttempts
|
|
919
981
|
});
|
|
920
982
|
results.push(result);
|
|
921
983
|
for (const turn of result.turns) {
|
|
@@ -1015,6 +1077,7 @@ var BenchmarkRunner = class {
|
|
|
1015
1077
|
promptVersion: this.config.promptVersion ?? "v1",
|
|
1016
1078
|
scenarioCount: toRun.length,
|
|
1017
1079
|
results,
|
|
1080
|
+
cost: costLedger.summary({ tags: costTags }),
|
|
1018
1081
|
summary: { overallAvg, byPersona, byDimension, weakest, strongest }
|
|
1019
1082
|
};
|
|
1020
1083
|
}
|
|
@@ -1621,11 +1684,15 @@ var AgentDriver = class {
|
|
|
1621
1684
|
client;
|
|
1622
1685
|
driverModel;
|
|
1623
1686
|
productContext;
|
|
1687
|
+
costLedger;
|
|
1688
|
+
tcloudMaximumAttempts;
|
|
1624
1689
|
constructor(tc, config) {
|
|
1625
1690
|
this.tc = tc;
|
|
1626
1691
|
this.client = config.client;
|
|
1627
1692
|
this.driverModel = config.driverModel ?? "claude-sonnet-4-6";
|
|
1628
1693
|
this.productContext = config.productContext ?? "";
|
|
1694
|
+
this.costLedger = config.costLedger ?? new CostLedger();
|
|
1695
|
+
this.tcloudMaximumAttempts = config.tcloudMaximumAttempts;
|
|
1629
1696
|
}
|
|
1630
1697
|
/**
|
|
1631
1698
|
* Run a persona through the product.
|
|
@@ -1634,6 +1701,7 @@ var AgentDriver = class {
|
|
|
1634
1701
|
* quality curve, and convergence curve.
|
|
1635
1702
|
*/
|
|
1636
1703
|
async run(persona) {
|
|
1704
|
+
const costTags = { driverRunId: globalThis.crypto.randomUUID() };
|
|
1637
1705
|
const email = `eval-driver-${Date.now()}@test.agent-eval.local`;
|
|
1638
1706
|
await this.client.signup(`Driver ${persona.role}`, email, "eval-driver-pass");
|
|
1639
1707
|
await this.client.login(email, "eval-driver-pass");
|
|
@@ -1648,7 +1716,12 @@ var AgentDriver = class {
|
|
|
1648
1716
|
let criteriaMetAtTurn = null;
|
|
1649
1717
|
for (let turn = 1; turn <= persona.maxTurns; turn++) {
|
|
1650
1718
|
const state = await metrics.getState();
|
|
1651
|
-
const userMessage = await this.decideNextMessage(
|
|
1719
|
+
const userMessage = await this.decideNextMessage(
|
|
1720
|
+
persona,
|
|
1721
|
+
state,
|
|
1722
|
+
conversationHistory,
|
|
1723
|
+
costTags
|
|
1724
|
+
);
|
|
1652
1725
|
if (userMessage === "DONE") {
|
|
1653
1726
|
completed = true;
|
|
1654
1727
|
turnsToCompletion = turn - 1;
|
|
@@ -1696,18 +1769,21 @@ var AgentDriver = class {
|
|
|
1696
1769
|
metrics: turnMetrics,
|
|
1697
1770
|
finalState,
|
|
1698
1771
|
convergenceCurve: convergence.getCurve(),
|
|
1699
|
-
totalCostUsd:
|
|
1772
|
+
totalCostUsd: this.costLedger.summary({ tags: costTags }).totalCostUsd,
|
|
1700
1773
|
finalQualityScore: null
|
|
1701
1774
|
};
|
|
1702
1775
|
}
|
|
1703
1776
|
/** Use the driver LLM to decide what the "user" says next */
|
|
1704
|
-
async decideNextMessage(persona, state, history) {
|
|
1777
|
+
async decideNextMessage(persona, state, history, costTags) {
|
|
1705
1778
|
return decideNextUserTurn(this.tc, {
|
|
1706
1779
|
persona,
|
|
1707
1780
|
state,
|
|
1708
1781
|
history,
|
|
1709
1782
|
productContext: this.productContext,
|
|
1710
|
-
model: this.driverModel
|
|
1783
|
+
model: this.driverModel,
|
|
1784
|
+
costLedger: this.costLedger,
|
|
1785
|
+
costTags,
|
|
1786
|
+
tcloudMaximumAttempts: this.tcloudMaximumAttempts
|
|
1711
1787
|
});
|
|
1712
1788
|
}
|
|
1713
1789
|
/** Handle pending approvals based on persona feedback patterns */
|
|
@@ -1816,7 +1892,7 @@ async function decideNextUserTurn(tc, opts) {
|
|
|
1816
1892
|
const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
|
|
1817
1893
|
const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet \u2014 this is the first message)";
|
|
1818
1894
|
const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
|
|
1819
|
-
const
|
|
1895
|
+
const request = {
|
|
1820
1896
|
model,
|
|
1821
1897
|
messages: [
|
|
1822
1898
|
{ role: "system", content: buildDriverSystemPrompt(persona, state, productContext) },
|
|
@@ -1831,7 +1907,19 @@ ${lastResponse}` : "No conversation yet. Send your opening message \u2014 in cha
|
|
|
1831
1907
|
],
|
|
1832
1908
|
temperature: 0.5,
|
|
1833
1909
|
maxTokens: 700
|
|
1910
|
+
};
|
|
1911
|
+
const paid = await (opts.costLedger ?? new CostLedger()).runPaidCall({
|
|
1912
|
+
channel: "driver",
|
|
1913
|
+
phase: "driver-turn",
|
|
1914
|
+
actor: "decideNextUserTurn",
|
|
1915
|
+
model,
|
|
1916
|
+
tags: opts.costTags,
|
|
1917
|
+
maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
|
|
1918
|
+
execute: () => tc.chat(request),
|
|
1919
|
+
receipt: (response) => costReceiptFromTCloud(response, model)
|
|
1834
1920
|
});
|
|
1921
|
+
if (!paid.succeeded) throw paid.error;
|
|
1922
|
+
const resp = paid.value;
|
|
1835
1923
|
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
1836
1924
|
return content.trim();
|
|
1837
1925
|
}
|
|
@@ -3794,7 +3882,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3794
3882
|
const dimAcc = {};
|
|
3795
3883
|
for (const d of dimensionKeys) dimAcc[d] = [];
|
|
3796
3884
|
let rationale = "";
|
|
3797
|
-
let costUsd = 0;
|
|
3798
3885
|
const seenCount = /* @__PURE__ */ new Map();
|
|
3799
3886
|
const keyFor = (model) => {
|
|
3800
3887
|
const n = (seenCount.get(model) ?? 0) + 1;
|
|
@@ -3802,7 +3889,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3802
3889
|
return n === 1 ? model : `${model}#${n}`;
|
|
3803
3890
|
};
|
|
3804
3891
|
for (const v of verdicts) {
|
|
3805
|
-
costUsd += v.costUsd ?? 0;
|
|
3806
3892
|
const key = keyFor(v.model);
|
|
3807
3893
|
if (!v.perDimension) {
|
|
3808
3894
|
failedJudges.push(key);
|
|
@@ -3844,7 +3930,6 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3844
3930
|
perJudge,
|
|
3845
3931
|
maxDisagreement,
|
|
3846
3932
|
failedJudges,
|
|
3847
|
-
costUsd,
|
|
3848
3933
|
rationale: rationale || "llm-judge",
|
|
3849
3934
|
verdicts: [...verdicts]
|
|
3850
3935
|
};
|
|
@@ -3924,37 +4009,92 @@ function ensembleJudge(opts) {
|
|
|
3924
4009
|
if (opts.crossFamily !== false) {
|
|
3925
4010
|
assertCrossFamily(opts.models);
|
|
3926
4011
|
}
|
|
3927
|
-
const
|
|
3928
|
-
|
|
3929
|
-
|
|
3930
|
-
|
|
3931
|
-
|
|
3932
|
-
|
|
3933
|
-
|
|
3934
|
-
|
|
4012
|
+
const declaredJudgeVersion = opts.judgeVersion?.trim();
|
|
4013
|
+
if (opts.judgeVersion !== void 0 && !declaredJudgeVersion) {
|
|
4014
|
+
throw new Error(`ensembleJudge '${opts.name}': judgeVersion must be non-empty when provided`);
|
|
4015
|
+
}
|
|
4016
|
+
const judgeVersion = declaredJudgeVersion ?? contentHash({
|
|
4017
|
+
kind: "ensembleJudge",
|
|
4018
|
+
models: opts.models,
|
|
4019
|
+
dimensions: opts.dimensions,
|
|
4020
|
+
weights: opts.weights ?? null,
|
|
4021
|
+
crossFamily: opts.crossFamily ?? true,
|
|
4022
|
+
maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge.toString() : opts.maximumCharge ?? null,
|
|
4023
|
+
retry: opts.retry ? {
|
|
4024
|
+
maxAttempts: opts.retry.maxAttempts ?? null,
|
|
4025
|
+
timeoutMs: opts.retry.timeoutMs ?? null,
|
|
4026
|
+
models: opts.retry.models ?? null,
|
|
4027
|
+
backoffMs: opts.retry.backoffMs?.toString() ?? null,
|
|
4028
|
+
isRetryable: opts.retry.isRetryable?.toString() ?? null
|
|
4029
|
+
} : null,
|
|
4030
|
+
scoreWith: opts.scoreWith.toString()
|
|
4031
|
+
});
|
|
4032
|
+
const directCostLedger = opts.costLedger ?? new CostLedger();
|
|
4033
|
+
const scoreOne = async (args) => {
|
|
4034
|
+
const outcome = await withJudgeRetry(
|
|
4035
|
+
async (model, retrySignal) => {
|
|
4036
|
+
const paid = await args.costLedger.runPaidCall({
|
|
4037
|
+
channel: "judge",
|
|
4038
|
+
phase: args.costPhase,
|
|
4039
|
+
actor: `${opts.name}.${model}`,
|
|
3935
4040
|
model,
|
|
3936
|
-
|
|
3937
|
-
|
|
3938
|
-
|
|
3939
|
-
|
|
3940
|
-
|
|
3941
|
-
|
|
3942
|
-
|
|
3943
|
-
|
|
3944
|
-
|
|
4041
|
+
maximumCharge: typeof opts.maximumCharge === "function" ? opts.maximumCharge(model) : opts.maximumCharge,
|
|
4042
|
+
tags: args.costTags,
|
|
4043
|
+
signal: AbortSignal.any([args.signal, retrySignal]),
|
|
4044
|
+
execute: (signal) => opts.scoreWith(model, { artifact: args.artifact, scenario: args.scenario, signal }),
|
|
4045
|
+
receipt: (verdict) => {
|
|
4046
|
+
const cachedTokens = verdict.usage?.cachedPromptTokens ?? 0;
|
|
4047
|
+
const usageUnknown = !verdict.usage || verdict.usage.captured === false;
|
|
4048
|
+
return {
|
|
4049
|
+
model: verdict.model,
|
|
4050
|
+
inputTokens: Math.max(0, (verdict.usage?.promptTokens ?? 0) - cachedTokens),
|
|
4051
|
+
outputTokens: verdict.usage?.completionTokens ?? 0,
|
|
4052
|
+
cachedTokens: cachedTokens > 0 ? cachedTokens : void 0,
|
|
4053
|
+
usageUnknown,
|
|
4054
|
+
...verdict.costUsd === void 0 ? {} : { actualCostUsd: verdict.costUsd }
|
|
4055
|
+
};
|
|
4056
|
+
},
|
|
4057
|
+
receiptFromError: (error) => opts.receiptFromError?.(error, model)
|
|
4058
|
+
});
|
|
4059
|
+
if (!paid.succeeded) throw paid.error;
|
|
4060
|
+
return paid.value;
|
|
4061
|
+
},
|
|
4062
|
+
opts.retry ? { ...opts.retry, models: [args.model] } : { maxAttempts: 1, models: [args.model], isRetryable: () => false }
|
|
4063
|
+
);
|
|
4064
|
+
if (!outcome.succeeded || outcome.value === null) {
|
|
3945
4065
|
return {
|
|
3946
|
-
model,
|
|
4066
|
+
model: args.model,
|
|
3947
4067
|
perDimension: null,
|
|
3948
|
-
rationale:
|
|
4068
|
+
rationale: outcome.error?.message ?? "judge failed"
|
|
3949
4069
|
};
|
|
3950
4070
|
}
|
|
4071
|
+
return outcome.value;
|
|
3951
4072
|
};
|
|
3952
4073
|
return {
|
|
3953
4074
|
name: opts.name,
|
|
3954
4075
|
dimensions: opts.dimensions.map((d) => ({ key: d, description: d })),
|
|
3955
|
-
|
|
3956
|
-
|
|
3957
|
-
|
|
4076
|
+
judgeVersion,
|
|
4077
|
+
async score({
|
|
4078
|
+
artifact,
|
|
4079
|
+
scenario,
|
|
4080
|
+
signal,
|
|
4081
|
+
costLedger,
|
|
4082
|
+
costPhase,
|
|
4083
|
+
costTags
|
|
4084
|
+
}) {
|
|
4085
|
+
const verdicts = await Promise.all(
|
|
4086
|
+
opts.models.map(
|
|
4087
|
+
(model) => scoreOne({
|
|
4088
|
+
model,
|
|
4089
|
+
artifact,
|
|
4090
|
+
scenario,
|
|
4091
|
+
signal,
|
|
4092
|
+
costLedger: costLedger ?? directCostLedger,
|
|
4093
|
+
costPhase: costPhase ?? "judge",
|
|
4094
|
+
costTags
|
|
4095
|
+
})
|
|
4096
|
+
)
|
|
4097
|
+
);
|
|
3958
4098
|
const agg = aggregateJudgeVerdicts(verdicts, opts.dimensions, opts.weights);
|
|
3959
4099
|
const score = {
|
|
3960
4100
|
dimensions: agg.perDimension,
|
|
@@ -4570,118 +4710,13 @@ function clampUnit(value) {
|
|
|
4570
4710
|
return Math.max(0, Math.min(1, value));
|
|
4571
4711
|
}
|
|
4572
4712
|
|
|
4573
|
-
// src/cost-ledger.ts
|
|
4574
|
-
function modelPriceKey(model) {
|
|
4575
|
-
return isModelPriced(model) ? model : null;
|
|
4576
|
-
}
|
|
4577
|
-
function costForUsage(model, usage) {
|
|
4578
|
-
assertNonNegative(usage.inputTokens, "inputTokens");
|
|
4579
|
-
assertNonNegative(usage.outputTokens, "outputTokens");
|
|
4580
|
-
if (usage.cachedTokens !== void 0) assertNonNegative(usage.cachedTokens, "cachedTokens");
|
|
4581
|
-
const pricing = resolveModelPricing(model);
|
|
4582
|
-
if (!pricing) return { costUsd: 0, costUnknown: true };
|
|
4583
|
-
const billedInput = usage.inputTokens + (usage.cachedTokens ?? 0);
|
|
4584
|
-
return { costUsd: estimateCost(billedInput, usage.outputTokens, model), costUnknown: false };
|
|
4585
|
-
}
|
|
4586
|
-
var CostLedger = class {
|
|
4587
|
-
entries = [];
|
|
4588
|
-
completedTasks = 0;
|
|
4589
|
-
/**
|
|
4590
|
-
* Record one LLM call. The cost is computed from pricing unless
|
|
4591
|
-
* `actualCostUsd` is supplied (a finite observed cost from the provider
|
|
4592
|
-
* response), in which case `costUnknown` is false regardless of pricing.
|
|
4593
|
-
*/
|
|
4594
|
-
record(input) {
|
|
4595
|
-
const { costUsd, costUnknown } = costForUsage(input.model, input.usage);
|
|
4596
|
-
const hasActual = typeof input.actualCostUsd === "number" && Number.isFinite(input.actualCostUsd);
|
|
4597
|
-
if (hasActual) assertNonNegative(input.actualCostUsd, "actualCostUsd");
|
|
4598
|
-
const entry = {
|
|
4599
|
-
model: input.model,
|
|
4600
|
-
channel: input.channel,
|
|
4601
|
-
inputTokens: input.usage.inputTokens,
|
|
4602
|
-
outputTokens: input.usage.outputTokens,
|
|
4603
|
-
cachedTokens: input.usage.cachedTokens,
|
|
4604
|
-
costUsd: hasActual ? input.actualCostUsd : costUsd,
|
|
4605
|
-
costUnknown: hasActual ? false : costUnknown,
|
|
4606
|
-
actualCostUsd: hasActual ? input.actualCostUsd : void 0,
|
|
4607
|
-
tags: input.tags,
|
|
4608
|
-
timestamp: input.timestamp ?? Date.now()
|
|
4609
|
-
};
|
|
4610
|
-
this.entries.push(entry);
|
|
4611
|
-
return entry;
|
|
4612
|
-
}
|
|
4613
|
-
/** Increment the completed-task counter (used for cost-per-completed-task). */
|
|
4614
|
-
markCompleted(count = 1) {
|
|
4615
|
-
if (!Number.isInteger(count) || count < 0) {
|
|
4616
|
-
throw new ValidationError(
|
|
4617
|
-
`CostLedger.markCompleted: count must be a non-negative integer, got ${count}`
|
|
4618
|
-
);
|
|
4619
|
-
}
|
|
4620
|
-
this.completedTasks += count;
|
|
4621
|
-
}
|
|
4622
|
-
list() {
|
|
4623
|
-
return [...this.entries];
|
|
4624
|
-
}
|
|
4625
|
-
summary() {
|
|
4626
|
-
const byChannel = /* @__PURE__ */ new Map();
|
|
4627
|
-
const unpriced = /* @__PURE__ */ new Set();
|
|
4628
|
-
let totalCost = 0;
|
|
4629
|
-
let inputTokens = 0;
|
|
4630
|
-
let outputTokens = 0;
|
|
4631
|
-
let cachedTokens = 0;
|
|
4632
|
-
for (const e of this.entries) {
|
|
4633
|
-
totalCost += e.costUsd;
|
|
4634
|
-
inputTokens += e.inputTokens;
|
|
4635
|
-
outputTokens += e.outputTokens;
|
|
4636
|
-
cachedTokens += e.cachedTokens ?? 0;
|
|
4637
|
-
if (e.costUnknown) unpriced.add(e.model);
|
|
4638
|
-
const roll = byChannel.get(e.channel) ?? {
|
|
4639
|
-
channel: e.channel,
|
|
4640
|
-
calls: 0,
|
|
4641
|
-
inputTokens: 0,
|
|
4642
|
-
outputTokens: 0,
|
|
4643
|
-
cachedTokens: 0,
|
|
4644
|
-
costUsd: 0,
|
|
4645
|
-
unpricedCalls: 0
|
|
4646
|
-
};
|
|
4647
|
-
roll.calls += 1;
|
|
4648
|
-
roll.inputTokens += e.inputTokens;
|
|
4649
|
-
roll.outputTokens += e.outputTokens;
|
|
4650
|
-
roll.cachedTokens += e.cachedTokens ?? 0;
|
|
4651
|
-
roll.costUsd += e.costUsd;
|
|
4652
|
-
if (e.costUnknown) roll.unpricedCalls += 1;
|
|
4653
|
-
byChannel.set(e.channel, roll);
|
|
4654
|
-
}
|
|
4655
|
-
return {
|
|
4656
|
-
totalCalls: this.entries.length,
|
|
4657
|
-
inputTokens,
|
|
4658
|
-
outputTokens,
|
|
4659
|
-
cachedTokens,
|
|
4660
|
-
totalCostUsd: totalCost,
|
|
4661
|
-
byChannel: [...byChannel.values()].sort((a, b) => a.channel.localeCompare(b.channel)),
|
|
4662
|
-
unpricedModels: [...unpriced].sort(),
|
|
4663
|
-
fullyPriced: unpriced.size === 0
|
|
4664
|
-
};
|
|
4665
|
-
}
|
|
4666
|
-
/** Total spend divided by completed tasks; null when nothing completed. */
|
|
4667
|
-
costPerCompletedTask() {
|
|
4668
|
-
if (this.completedTasks === 0) return null;
|
|
4669
|
-
return this.summary().totalCostUsd / this.completedTasks;
|
|
4670
|
-
}
|
|
4671
|
-
};
|
|
4672
|
-
function assertNonNegative(n, name) {
|
|
4673
|
-
if (!Number.isFinite(n) || n < 0) {
|
|
4674
|
-
throw new ValidationError(`CostLedger: ${name} must be a non-negative finite number, got ${n}`);
|
|
4675
|
-
}
|
|
4676
|
-
}
|
|
4677
|
-
|
|
4678
4713
|
// src/cost-tracker.ts
|
|
4679
4714
|
var CostTracker = class {
|
|
4680
4715
|
byScenario = /* @__PURE__ */ new Map();
|
|
4681
4716
|
record(entry) {
|
|
4682
4717
|
const full = { timestamp: entry.timestamp ?? Date.now(), ...entry };
|
|
4683
|
-
|
|
4684
|
-
|
|
4718
|
+
assertNonNegative(full.inputTokens, "inputTokens");
|
|
4719
|
+
assertNonNegative(full.outputTokens, "outputTokens");
|
|
4685
4720
|
let bucket = this.byScenario.get(full.scenarioId);
|
|
4686
4721
|
if (!bucket) {
|
|
4687
4722
|
bucket = {
|
|
@@ -4760,7 +4795,7 @@ function costFor(entry) {
|
|
|
4760
4795
|
}
|
|
4761
4796
|
return estimateCost(entry.inputTokens, entry.outputTokens, entry.model);
|
|
4762
4797
|
}
|
|
4763
|
-
function
|
|
4798
|
+
function assertNonNegative(n, name) {
|
|
4764
4799
|
if (!Number.isFinite(n) || n < 0) {
|
|
4765
4800
|
throw new Error(`CostTracker: ${name} must be a non-negative finite number, got ${n}`);
|
|
4766
4801
|
}
|
|
@@ -7986,6 +8021,7 @@ function flowLayer(input) {
|
|
|
7986
8021
|
var INTENT_MATCH_JUDGE_VERSION = "intent-match-judge-v1-2026-04-24";
|
|
7987
8022
|
var DEFAULT_MODEL = "claude-sonnet-4-6";
|
|
7988
8023
|
var DEFAULT_TIMEOUT = 3e5;
|
|
8024
|
+
var DEFAULT_MAX_TOKENS = 800;
|
|
7989
8025
|
var DEFAULT_MAX_SOURCE = 25e3;
|
|
7990
8026
|
var DEFAULT_MAX_PER_FILE = 12e3;
|
|
7991
8027
|
var DEFAULT_MAX_HTML = 2e4;
|
|
@@ -8055,10 +8091,14 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8055
8091
|
const opts = {
|
|
8056
8092
|
model: options.model ?? DEFAULT_MODEL,
|
|
8057
8093
|
timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
|
|
8094
|
+
maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
8058
8095
|
maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
|
|
8059
8096
|
maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
|
|
8060
8097
|
maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
|
|
8061
|
-
llm: options.llm ?? {}
|
|
8098
|
+
llm: options.llm ?? {},
|
|
8099
|
+
costLedger: options.costLedger ?? new CostLedger(),
|
|
8100
|
+
costPhase: options.costPhase ?? "judge.intent-match",
|
|
8101
|
+
signal: options.signal ?? new AbortController().signal
|
|
8062
8102
|
};
|
|
8063
8103
|
if (input.sourceFiles.length === 0 && !input.servedHtml) {
|
|
8064
8104
|
return {
|
|
@@ -8072,23 +8112,40 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8072
8112
|
error: "no input artifact"
|
|
8073
8113
|
};
|
|
8074
8114
|
}
|
|
8115
|
+
let receipt;
|
|
8075
8116
|
try {
|
|
8076
|
-
const
|
|
8077
|
-
|
|
8078
|
-
|
|
8079
|
-
|
|
8080
|
-
|
|
8081
|
-
|
|
8082
|
-
|
|
8083
|
-
|
|
8084
|
-
|
|
8085
|
-
|
|
8086
|
-
|
|
8087
|
-
|
|
8088
|
-
|
|
8089
|
-
|
|
8090
|
-
|
|
8091
|
-
|
|
8117
|
+
const request = {
|
|
8118
|
+
model: opts.model,
|
|
8119
|
+
messages: [
|
|
8120
|
+
{
|
|
8121
|
+
role: "system",
|
|
8122
|
+
content: "You are a holistic code reviewer answering one question: did the agent build the right app for the user. Return strict JSON. No prose outside."
|
|
8123
|
+
},
|
|
8124
|
+
{ role: "user", content: buildPrompt(input, opts) }
|
|
8125
|
+
],
|
|
8126
|
+
jsonSchema: { name: "intent_match_judge", schema: INTENT_SCHEMA },
|
|
8127
|
+
temperature: 0,
|
|
8128
|
+
maxTokens: opts.maxTokens,
|
|
8129
|
+
timeoutMs: opts.timeoutMs
|
|
8130
|
+
};
|
|
8131
|
+
const paid = await opts.costLedger.runPaidCall({
|
|
8132
|
+
channel: "judge",
|
|
8133
|
+
phase: opts.costPhase,
|
|
8134
|
+
actor: "intent-match",
|
|
8135
|
+
model: opts.model,
|
|
8136
|
+
maximumCharge: maximumChargeForLlmRequest(request, opts.llm),
|
|
8137
|
+
signal: opts.signal,
|
|
8138
|
+
execute: (signal, callId) => callLlmJson(request, {
|
|
8139
|
+
...opts.llm,
|
|
8140
|
+
signal,
|
|
8141
|
+
idempotencyKey: callId
|
|
8142
|
+
}),
|
|
8143
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
8144
|
+
receiptFromError: costReceiptFromLlmError
|
|
8145
|
+
});
|
|
8146
|
+
receipt = paid.receipt;
|
|
8147
|
+
if (!paid.succeeded) throw paid.error;
|
|
8148
|
+
const { value } = paid.value;
|
|
8092
8149
|
const score = Math.max(0, Math.min(1, Number(value?.score ?? 0)));
|
|
8093
8150
|
return {
|
|
8094
8151
|
kind: "intent-match",
|
|
@@ -8096,7 +8153,7 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8096
8153
|
score: Number(score.toFixed(3)),
|
|
8097
8154
|
evidence: String(value?.evidence ?? "").slice(0, 400),
|
|
8098
8155
|
durationMs: Date.now() - start,
|
|
8099
|
-
costUsd:
|
|
8156
|
+
costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
|
|
8100
8157
|
available: true
|
|
8101
8158
|
};
|
|
8102
8159
|
} catch (err) {
|
|
@@ -8106,7 +8163,7 @@ async function runIntentMatchJudge(input, options = {}) {
|
|
|
8106
8163
|
score: 0,
|
|
8107
8164
|
evidence: "",
|
|
8108
8165
|
durationMs: Date.now() - start,
|
|
8109
|
-
costUsd: null,
|
|
8166
|
+
costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
|
|
8110
8167
|
available: false,
|
|
8111
8168
|
error: err instanceof Error ? err.message : String(err)
|
|
8112
8169
|
};
|
|
@@ -9146,23 +9203,42 @@ function createDefaultReviewer(options) {
|
|
|
9146
9203
|
};
|
|
9147
9204
|
const promptBuilder = options.promptBuilder ?? buildReviewerPrompt;
|
|
9148
9205
|
const timeoutMs = options.timeoutMs ?? 3e5;
|
|
9206
|
+
const maxTokens = options.maxTokens ?? 4e3;
|
|
9207
|
+
const costLedger = options.costLedger ?? new CostLedger();
|
|
9149
9208
|
return async (input) => {
|
|
9150
9209
|
const start = Date.now();
|
|
9151
9210
|
const { system, user } = promptBuilder(input);
|
|
9211
|
+
let receipt;
|
|
9152
9212
|
try {
|
|
9153
|
-
const
|
|
9154
|
-
|
|
9155
|
-
|
|
9156
|
-
|
|
9157
|
-
|
|
9158
|
-
|
|
9159
|
-
|
|
9160
|
-
|
|
9161
|
-
|
|
9162
|
-
|
|
9163
|
-
|
|
9164
|
-
|
|
9165
|
-
|
|
9213
|
+
const request = {
|
|
9214
|
+
model: options.model,
|
|
9215
|
+
messages: [
|
|
9216
|
+
{ role: "system", content: system },
|
|
9217
|
+
{ role: "user", content: user }
|
|
9218
|
+
],
|
|
9219
|
+
jsonSchema: { name: "reviewer_output", schema: REVIEWER_SCHEMA },
|
|
9220
|
+
temperature: 0,
|
|
9221
|
+
maxTokens,
|
|
9222
|
+
timeoutMs
|
|
9223
|
+
};
|
|
9224
|
+
const paid = await costLedger.runPaidCall({
|
|
9225
|
+
channel: "analyst",
|
|
9226
|
+
phase: options.costPhase ?? "review",
|
|
9227
|
+
actor: "default-reviewer",
|
|
9228
|
+
model: options.model,
|
|
9229
|
+
maximumCharge: maximumChargeForLlmRequest(request, options.llm),
|
|
9230
|
+
signal: options.signal,
|
|
9231
|
+
execute: (signal, callId) => callLlmJson(request, {
|
|
9232
|
+
...options.llm,
|
|
9233
|
+
signal,
|
|
9234
|
+
idempotencyKey: callId
|
|
9235
|
+
}),
|
|
9236
|
+
receipt: ({ result }) => costReceiptFromLlm(result),
|
|
9237
|
+
receiptFromError: costReceiptFromLlmError
|
|
9238
|
+
});
|
|
9239
|
+
receipt = paid.receipt;
|
|
9240
|
+
if (!paid.succeeded) throw paid.error;
|
|
9241
|
+
const { value } = paid.value;
|
|
9166
9242
|
return {
|
|
9167
9243
|
shot: input.shot,
|
|
9168
9244
|
observations: String(value.observations ?? softFail2.observations),
|
|
@@ -9170,7 +9246,7 @@ function createDefaultReviewer(options) {
|
|
|
9170
9246
|
nextShotInstruction: String(value.nextShotInstruction ?? softFail2.nextShotInstruction),
|
|
9171
9247
|
shouldContinue: Boolean(value.shouldContinue),
|
|
9172
9248
|
confidence: Math.max(0, Math.min(1, Number(value.confidence ?? softFail2.confidence))),
|
|
9173
|
-
costUsd:
|
|
9249
|
+
costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
|
|
9174
9250
|
durationMs: Date.now() - start,
|
|
9175
9251
|
available: true
|
|
9176
9252
|
};
|
|
@@ -9182,7 +9258,7 @@ function createDefaultReviewer(options) {
|
|
|
9182
9258
|
nextShotInstruction: softFail2.nextShotInstruction,
|
|
9183
9259
|
shouldContinue: softFail2.shouldContinue,
|
|
9184
9260
|
confidence: softFail2.confidence,
|
|
9185
|
-
costUsd: null,
|
|
9261
|
+
costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
|
|
9186
9262
|
durationMs: Date.now() - start,
|
|
9187
9263
|
available: false,
|
|
9188
9264
|
error: err instanceof Error ? err.message : String(err)
|
|
@@ -10262,18 +10338,23 @@ async function runDistillation(opts) {
|
|
|
10262
10338
|
input: scenario.input,
|
|
10263
10339
|
scenarioId: scenario.id
|
|
10264
10340
|
});
|
|
10265
|
-
const
|
|
10266
|
-
|
|
10267
|
-
|
|
10268
|
-
|
|
10269
|
-
|
|
10270
|
-
|
|
10271
|
-
|
|
10272
|
-
|
|
10273
|
-
|
|
10274
|
-
|
|
10275
|
-
|
|
10276
|
-
|
|
10341
|
+
const request = {
|
|
10342
|
+
model: opts.studentModel,
|
|
10343
|
+
messages: prompt,
|
|
10344
|
+
jsonMode: true,
|
|
10345
|
+
temperature: studentTemperature,
|
|
10346
|
+
maxTokens: studentMaxTokens
|
|
10347
|
+
};
|
|
10348
|
+
const paid = await ctx.cost.runPaidCall({
|
|
10349
|
+
actor: "distillation-student",
|
|
10350
|
+
model: opts.studentModel,
|
|
10351
|
+
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maxRetries: chat.maximumAttempts }),
|
|
10352
|
+
execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
|
|
10353
|
+
receipt: costReceiptFromLlm,
|
|
10354
|
+
receiptFromError: costReceiptFromLlmError
|
|
10355
|
+
});
|
|
10356
|
+
if (!paid.succeeded) throw paid.error;
|
|
10357
|
+
return parse(paid.value.content, scenario.id);
|
|
10277
10358
|
}
|
|
10278
10359
|
});
|
|
10279
10360
|
const winnerPrompt = typeof loop.winnerSurface === "string" ? loop.winnerSurface : opts.baselinePrompt;
|
|
@@ -10285,14 +10366,6 @@ async function runDistillation(opts) {
|
|
|
10285
10366
|
holdoutAgreement: { baseline, winner, delta: winner - baseline }
|
|
10286
10367
|
};
|
|
10287
10368
|
}
|
|
10288
|
-
function reportUsage(cost, response) {
|
|
10289
|
-
if (typeof response.costUsd === "number") cost.observe(response.costUsd, "distillation-student");
|
|
10290
|
-
cost.observeTokens({
|
|
10291
|
-
input: response.usage.promptTokens,
|
|
10292
|
-
output: response.usage.completionTokens,
|
|
10293
|
-
cached: response.usage.cachedPromptTokens
|
|
10294
|
-
});
|
|
10295
|
-
}
|
|
10296
10369
|
var DEFAULT_MUTATION_PRIMITIVES2 = [
|
|
10297
10370
|
"Add an explicit output-schema instruction so the model emits exactly the gold label fields as JSON.",
|
|
10298
10371
|
"Add a one-line decision rule for each verdict field the student keeps getting wrong.",
|
|
@@ -11455,7 +11528,13 @@ export {
|
|
|
11455
11528
|
CaptureIntegrityError,
|
|
11456
11529
|
ConfigError,
|
|
11457
11530
|
ConvergenceTracker,
|
|
11531
|
+
CostAccountingIncompleteError,
|
|
11532
|
+
CostCallConflictError,
|
|
11533
|
+
CostCeilingReachedError,
|
|
11458
11534
|
CostLedger,
|
|
11535
|
+
CostLedgerPersistenceError,
|
|
11536
|
+
CostReceiptCaptureError,
|
|
11537
|
+
CostReservationExceededError,
|
|
11459
11538
|
CostTracker,
|
|
11460
11539
|
CrossFamilyError,
|
|
11461
11540
|
DEFAULT_AGENT_SLOS,
|
|
@@ -11490,6 +11569,7 @@ export {
|
|
|
11490
11569
|
HoldoutAuditor,
|
|
11491
11570
|
HoldoutLockedError,
|
|
11492
11571
|
IMPROVEMENT_KIND_SPEC,
|
|
11572
|
+
INPUT_VALUE,
|
|
11493
11573
|
INTENT_MATCH_JUDGE_VERSION,
|
|
11494
11574
|
InMemoryFeedbackTrajectoryStore,
|
|
11495
11575
|
InMemoryRawProviderSink,
|
|
@@ -11512,6 +11592,7 @@ export {
|
|
|
11512
11592
|
LLM_OUTPUT_TOKEN_ATTR_KEYS,
|
|
11513
11593
|
LlmCallError,
|
|
11514
11594
|
LlmClient,
|
|
11595
|
+
LlmResponseError,
|
|
11515
11596
|
LlmRouteAssertionError,
|
|
11516
11597
|
LockedJsonlAppender,
|
|
11517
11598
|
MODEL_PRICING,
|
|
@@ -11524,6 +11605,7 @@ export {
|
|
|
11524
11605
|
NotFoundError,
|
|
11525
11606
|
OPENINFERENCE_SPAN_KIND,
|
|
11526
11607
|
OTEL_AGENT_EVAL_SCOPE,
|
|
11608
|
+
OUTPUT_VALUE,
|
|
11527
11609
|
OtlpFileTraceStore,
|
|
11528
11610
|
POLICY_EDIT_AXES,
|
|
11529
11611
|
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
@@ -11533,6 +11615,8 @@ export {
|
|
|
11533
11615
|
ProductClient,
|
|
11534
11616
|
PromptRegistry,
|
|
11535
11617
|
REDACTION_VERSION,
|
|
11618
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
11619
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
11536
11620
|
RESEARCH_REPORT_HARD_PAIR_FLOOR,
|
|
11537
11621
|
ReplayCache,
|
|
11538
11622
|
ReplayCacheMissError,
|
|
@@ -11550,6 +11634,8 @@ export {
|
|
|
11550
11634
|
SkillUsageAnalyst,
|
|
11551
11635
|
SpanNotFoundError,
|
|
11552
11636
|
SubprocessSandboxDriver,
|
|
11637
|
+
TOOL_ARGS_CAPTURED,
|
|
11638
|
+
TOOL_LATENCY_MS,
|
|
11553
11639
|
TOOL_NAME,
|
|
11554
11640
|
TOOL_NAME_ATTR_KEYS,
|
|
11555
11641
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
@@ -11586,6 +11672,7 @@ export {
|
|
|
11586
11672
|
analyzeTraces,
|
|
11587
11673
|
appendScorecard,
|
|
11588
11674
|
applyPolicyEditToSurface,
|
|
11675
|
+
applyToolSpanOtlpAttributes,
|
|
11589
11676
|
argHash,
|
|
11590
11677
|
asNumber,
|
|
11591
11678
|
asString,
|
|
@@ -11678,6 +11765,8 @@ export {
|
|
|
11678
11765
|
corpusInterRaterAgreement,
|
|
11679
11766
|
corpusInterRaterAgreementFromJudgeScores,
|
|
11680
11767
|
costForUsage,
|
|
11768
|
+
costReceiptFromLlm,
|
|
11769
|
+
costReceiptFromLlmError,
|
|
11681
11770
|
costReport,
|
|
11682
11771
|
createAnalystAi,
|
|
11683
11772
|
createAntiSlopJudge,
|
|
@@ -11691,6 +11780,7 @@ export {
|
|
|
11691
11780
|
createLlmReviewer,
|
|
11692
11781
|
createOtelExporter,
|
|
11693
11782
|
createOtelTracingStore,
|
|
11783
|
+
createReferenceEquivalenceJudge,
|
|
11694
11784
|
createReplayFetch,
|
|
11695
11785
|
createSandboxPool,
|
|
11696
11786
|
createSemanticConceptJudge,
|
|
@@ -11781,6 +11871,7 @@ export {
|
|
|
11781
11871
|
groupBy,
|
|
11782
11872
|
groupRunsByAgentProfileCell,
|
|
11783
11873
|
harnessAxisOf,
|
|
11874
|
+
hasCapturedToolArgs,
|
|
11784
11875
|
hashContent,
|
|
11785
11876
|
hashJson,
|
|
11786
11877
|
hashScenarios,
|
|
@@ -11840,6 +11931,7 @@ export {
|
|
|
11840
11931
|
mannWhitneyU,
|
|
11841
11932
|
matchGoldens,
|
|
11842
11933
|
matchSpan,
|
|
11934
|
+
maximumChargeForLlmRequest,
|
|
11843
11935
|
mcnemar,
|
|
11844
11936
|
mcnemarPower,
|
|
11845
11937
|
mcnemarRequiredN,
|
|
@@ -11955,6 +12047,7 @@ export {
|
|
|
11955
12047
|
runProposeReview,
|
|
11956
12048
|
runProposeReviewAsControlLoop,
|
|
11957
12049
|
runRecordToProductBenchmarkRecord,
|
|
12050
|
+
runReferenceEquivalenceJudge,
|
|
11958
12051
|
runReferenceReplay,
|
|
11959
12052
|
runScore,
|
|
11960
12053
|
runSelfPlay,
|