@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -9,11 +9,15 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-M4YBQKIJ.js";
|
|
13
13
|
import {
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
14
|
+
ATIF_SCHEMA_VERSION,
|
|
15
|
+
HARBOR_IMPORT_GAP,
|
|
16
|
+
fromHarborTrajectory,
|
|
17
|
+
relabelImportedSplit,
|
|
18
|
+
toHarborTrajectories,
|
|
19
|
+
toHarborTrajectory
|
|
20
|
+
} from "./chunk-RXHCETDZ.js";
|
|
17
21
|
import {
|
|
18
22
|
SUPERVISOR_RUN_SCHEMA,
|
|
19
23
|
analyzeSupervisorRun,
|
|
@@ -27,25 +31,14 @@ import {
|
|
|
27
31
|
showMeasured,
|
|
28
32
|
supervisorRunRolloutLines,
|
|
29
33
|
writeSupervisorRunReport
|
|
30
|
-
} from "./chunk-
|
|
31
|
-
import "./chunk-
|
|
32
|
-
import
|
|
33
|
-
toJsonl,
|
|
34
|
-
toRewardRows,
|
|
35
|
-
toSftRows
|
|
36
|
-
} from "./chunk-EJGRPCO3.js";
|
|
37
|
-
import {
|
|
38
|
-
ROLLOUT_SCHEMA,
|
|
39
|
-
assertRolloutLine,
|
|
40
|
-
isRolloutLine,
|
|
41
|
-
isTrainableSplit,
|
|
42
|
-
validateRolloutLine
|
|
43
|
-
} from "./chunk-UWZZKKU7.js";
|
|
34
|
+
} from "./chunk-X4YIBDER.js";
|
|
35
|
+
import "./chunk-HPWUNB47.js";
|
|
36
|
+
import "./chunk-3OCR4R5I.js";
|
|
44
37
|
import {
|
|
45
38
|
BENCHMARK_SPLIT_SEED,
|
|
46
39
|
benchmarks_exports,
|
|
47
40
|
deterministicSplit
|
|
48
|
-
} from "./chunk-
|
|
41
|
+
} from "./chunk-IYCLP2N2.js";
|
|
49
42
|
import {
|
|
50
43
|
DEFAULT_RULES,
|
|
51
44
|
classifyFailure,
|
|
@@ -53,10 +46,7 @@ import {
|
|
|
53
46
|
computeToolUseMetrics,
|
|
54
47
|
iqr,
|
|
55
48
|
welchsTTest
|
|
56
|
-
} from "./chunk-
|
|
57
|
-
import {
|
|
58
|
-
buildTrajectory
|
|
59
|
-
} from "./chunk-RZTMDUO7.js";
|
|
49
|
+
} from "./chunk-FXTVJPYD.js";
|
|
60
50
|
import {
|
|
61
51
|
analyzeSeries
|
|
62
52
|
} from "./chunk-BOD4O7OF.js";
|
|
@@ -84,7 +74,7 @@ import {
|
|
|
84
74
|
harnessAxisOf,
|
|
85
75
|
parseCorrectnessResponse,
|
|
86
76
|
verifyCompletion
|
|
87
|
-
} from "./chunk-
|
|
77
|
+
} from "./chunk-QB6BDBP2.js";
|
|
88
78
|
import {
|
|
89
79
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
90
80
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -93,20 +83,12 @@ import {
|
|
|
93
83
|
JudgeParseError,
|
|
94
84
|
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
95
85
|
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
96
|
-
adversarialJudge,
|
|
97
86
|
buildReflectionPrompt,
|
|
98
|
-
codeExecutionJudge,
|
|
99
|
-
coherenceJudge,
|
|
100
|
-
costReceiptFromTCloud,
|
|
101
|
-
createCustomJudge,
|
|
102
|
-
createDomainExpertJudge,
|
|
103
87
|
createReferenceEquivalenceJudge,
|
|
104
88
|
crowdingDistance,
|
|
105
|
-
defaultJudges,
|
|
106
89
|
dominates,
|
|
107
90
|
hashScenarios,
|
|
108
91
|
llmJudge,
|
|
109
|
-
maximumChargeForTCloudRequest,
|
|
110
92
|
paretoFrontier,
|
|
111
93
|
paretoFrontierWithCrowding,
|
|
112
94
|
parseReflectionResponse,
|
|
@@ -118,7 +100,7 @@ import {
|
|
|
118
100
|
scoreRedTeamOutput,
|
|
119
101
|
surfaceContentHash,
|
|
120
102
|
toolNamesForRun
|
|
121
|
-
} from "./chunk-
|
|
103
|
+
} from "./chunk-2QU3YOPR.js";
|
|
122
104
|
import {
|
|
123
105
|
BackendIntegrityError,
|
|
124
106
|
assertRealAgentReceipts,
|
|
@@ -130,7 +112,7 @@ import {
|
|
|
130
112
|
inMemoryVerdictCache,
|
|
131
113
|
summarizeAgentReceiptIntegrity,
|
|
132
114
|
summarizeBackendIntegrity
|
|
133
|
-
} from "./chunk-
|
|
115
|
+
} from "./chunk-C6LXANRU.js";
|
|
134
116
|
import {
|
|
135
117
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
136
118
|
FindingsStore,
|
|
@@ -143,7 +125,7 @@ import {
|
|
|
143
125
|
defaultIsMaterial,
|
|
144
126
|
diffFindings,
|
|
145
127
|
runSemanticConceptJudge
|
|
146
|
-
} from "./chunk-
|
|
128
|
+
} from "./chunk-DODXQREJ.js";
|
|
147
129
|
import {
|
|
148
130
|
AnalystRegistry,
|
|
149
131
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -160,7 +142,7 @@ import {
|
|
|
160
142
|
makeFinding,
|
|
161
143
|
renderPriorFindings,
|
|
162
144
|
renderUpstreamFindings
|
|
163
|
-
} from "./chunk-
|
|
145
|
+
} from "./chunk-BSO5JDQH.js";
|
|
164
146
|
import "./chunk-HHWE3POT.js";
|
|
165
147
|
import {
|
|
166
148
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
@@ -189,19 +171,39 @@ import {
|
|
|
189
171
|
stopOnNoProgress,
|
|
190
172
|
stopOnRepeatedAction,
|
|
191
173
|
subjectiveEval
|
|
192
|
-
} from "./chunk-
|
|
174
|
+
} from "./chunk-WVATSFCP.js";
|
|
193
175
|
import {
|
|
194
176
|
assertReleaseConfidence,
|
|
195
177
|
bootstrapCi,
|
|
196
178
|
evaluateReleaseConfidence,
|
|
197
179
|
judgeReplayGate,
|
|
198
180
|
renderReleaseReport
|
|
199
|
-
} from "./chunk-
|
|
181
|
+
} from "./chunk-JQSF5DQT.js";
|
|
200
182
|
import {
|
|
201
183
|
runEvalCampaign
|
|
202
|
-
} from "./chunk-
|
|
203
|
-
import
|
|
204
|
-
|
|
184
|
+
} from "./chunk-YQN4ICPP.js";
|
|
185
|
+
import {
|
|
186
|
+
mintRolloutRows
|
|
187
|
+
} from "./chunk-H23X7XKK.js";
|
|
188
|
+
import {
|
|
189
|
+
toJsonl,
|
|
190
|
+
toRewardRows,
|
|
191
|
+
toSftRows
|
|
192
|
+
} from "./chunk-OWN5NPMC.js";
|
|
193
|
+
import {
|
|
194
|
+
ROLLOUT_SCHEMA,
|
|
195
|
+
assertMinted,
|
|
196
|
+
assertMintedLines,
|
|
197
|
+
assertRolloutLine,
|
|
198
|
+
isRolloutLine,
|
|
199
|
+
isTrainableSplit,
|
|
200
|
+
validateRolloutLine
|
|
201
|
+
} from "./chunk-PC5DOSM7.js";
|
|
202
|
+
import {
|
|
203
|
+
buildTrajectory
|
|
204
|
+
} from "./chunk-RZTMDUO7.js";
|
|
205
|
+
import "./chunk-EG66UGL4.js";
|
|
206
|
+
import "./chunk-E7QXT7SX.js";
|
|
205
207
|
import {
|
|
206
208
|
LlmCallError,
|
|
207
209
|
LlmClient,
|
|
@@ -217,7 +219,7 @@ import {
|
|
|
217
219
|
maximumChargeForLlmRequest,
|
|
218
220
|
probeLlm,
|
|
219
221
|
stripFencedJson
|
|
220
|
-
} from "./chunk-
|
|
222
|
+
} from "./chunk-SFLLL76A.js";
|
|
221
223
|
import {
|
|
222
224
|
evaluateInterimReleaseConfidence,
|
|
223
225
|
pairedEvalueSequence
|
|
@@ -228,12 +230,12 @@ import {
|
|
|
228
230
|
paretoChart,
|
|
229
231
|
researchReport,
|
|
230
232
|
summaryTable
|
|
231
|
-
} from "./chunk-
|
|
233
|
+
} from "./chunk-TJVT4QFF.js";
|
|
232
234
|
import {
|
|
233
235
|
comparePairedArms,
|
|
234
236
|
pairArms,
|
|
235
237
|
pairRunRecords
|
|
236
|
-
} from "./chunk-
|
|
238
|
+
} from "./chunk-7FO3TNPI.js";
|
|
237
239
|
import {
|
|
238
240
|
benjaminiHochberg,
|
|
239
241
|
bonferroni,
|
|
@@ -275,7 +277,7 @@ import {
|
|
|
275
277
|
weightedMean,
|
|
276
278
|
wilcoxonSignedRank,
|
|
277
279
|
wilson
|
|
278
|
-
} from "./chunk-
|
|
280
|
+
} from "./chunk-ZHTZ4EYI.js";
|
|
279
281
|
import {
|
|
280
282
|
CostAccountingIncompleteError,
|
|
281
283
|
CostCallConflictError,
|
|
@@ -287,7 +289,7 @@ import {
|
|
|
287
289
|
costForTokenPricing,
|
|
288
290
|
costForUsage,
|
|
289
291
|
modelPriceKey
|
|
290
|
-
} from "./chunk-
|
|
292
|
+
} from "./chunk-VCZ5FQYW.js";
|
|
291
293
|
import {
|
|
292
294
|
MODEL_PRICING,
|
|
293
295
|
MetricsCollector,
|
|
@@ -298,8 +300,6 @@ import {
|
|
|
298
300
|
resolveModelPricing
|
|
299
301
|
} from "./chunk-VI2UW6B6.js";
|
|
300
302
|
import {
|
|
301
|
-
FileSystemTraceStore,
|
|
302
|
-
InMemoryTraceStore,
|
|
303
303
|
OTEL_AGENT_EVAL_SCOPE,
|
|
304
304
|
ReplayCache,
|
|
305
305
|
ReplayCacheMissError,
|
|
@@ -326,7 +326,7 @@ import {
|
|
|
326
326
|
scoreTraceInsightReadiness,
|
|
327
327
|
tokenizeDomainWords,
|
|
328
328
|
traceAnalystOnRunComplete
|
|
329
|
-
} from "./chunk-
|
|
329
|
+
} from "./chunk-U4L7JRPZ.js";
|
|
330
330
|
import "./chunk-7ZZMD7UK.js";
|
|
331
331
|
import {
|
|
332
332
|
extractUsage,
|
|
@@ -377,10 +377,12 @@ import {
|
|
|
377
377
|
traceSpanKindToOpenInferenceKind
|
|
378
378
|
} from "./chunk-P6FYH6K4.js";
|
|
379
379
|
import {
|
|
380
|
+
FileSystemTraceStore,
|
|
381
|
+
InMemoryTraceStore,
|
|
380
382
|
RunIntegrityError,
|
|
381
383
|
assertRunCaptured,
|
|
382
384
|
throwIfRunIncomplete
|
|
383
|
-
} from "./chunk-
|
|
385
|
+
} from "./chunk-U4PHLT2N.js";
|
|
384
386
|
import {
|
|
385
387
|
FileSystemRawProviderSink,
|
|
386
388
|
InMemoryRawProviderSink,
|
|
@@ -417,7 +419,7 @@ import {
|
|
|
417
419
|
validateRunRecord,
|
|
418
420
|
verifyAgentProfileCell,
|
|
419
421
|
verifyManifest
|
|
420
|
-
} from "./chunk-
|
|
422
|
+
} from "./chunk-56TAVBOK.js";
|
|
421
423
|
import {
|
|
422
424
|
FAILURE_CLASSES,
|
|
423
425
|
TRACE_SCHEMA_VERSION,
|
|
@@ -427,6 +429,14 @@ import {
|
|
|
427
429
|
isSandboxSpan,
|
|
428
430
|
isToolSpan
|
|
429
431
|
} from "./chunk-MA6HLL3S.js";
|
|
432
|
+
import {
|
|
433
|
+
isRealnessGated,
|
|
434
|
+
observedScore,
|
|
435
|
+
observedSplitScore,
|
|
436
|
+
scoreOrigin,
|
|
437
|
+
trainingReward,
|
|
438
|
+
trainingScore
|
|
439
|
+
} from "./chunk-OIUOT4QD.js";
|
|
430
440
|
import {
|
|
431
441
|
AgentEvalError,
|
|
432
442
|
CaptureIntegrityError,
|
|
@@ -764,15 +774,11 @@ function renderCommitMessage(input) {
|
|
|
764
774
|
}
|
|
765
775
|
|
|
766
776
|
// src/executor.ts
|
|
767
|
-
function describeShape(content, message) {
|
|
768
|
-
if (message === void 0 || message === null) return "no choices[0].message";
|
|
769
|
-
return `message.content of type ${content === null ? "null" : typeof content}`;
|
|
770
|
-
}
|
|
771
777
|
function errMessage(err) {
|
|
772
778
|
if (err instanceof Error) return `${err.name}: ${err.message}`;
|
|
773
779
|
return String(err);
|
|
774
780
|
}
|
|
775
|
-
async function executeScenario(
|
|
781
|
+
async function executeScenario(chat, scenario, config) {
|
|
776
782
|
const startTime = Date.now();
|
|
777
783
|
const model = config.model ?? "gpt-4o";
|
|
778
784
|
const costLedger = config.costLedger ?? new CostLedger();
|
|
@@ -803,19 +809,18 @@ async function executeScenario(tc, scenario, config) {
|
|
|
803
809
|
phase: config.costPhase ?? "benchmark.agent",
|
|
804
810
|
actor: "scenario-agent",
|
|
805
811
|
model,
|
|
806
|
-
maximumCharge:
|
|
812
|
+
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
|
|
807
813
|
tags: costTags,
|
|
808
814
|
signal: config.signal,
|
|
809
|
-
execute: () =>
|
|
810
|
-
receipt:
|
|
815
|
+
execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
|
|
816
|
+
receipt: costReceiptFromLlm,
|
|
817
|
+
receiptFromError: costReceiptFromLlmError
|
|
811
818
|
});
|
|
812
819
|
if (!paid.succeeded) throw paid.error;
|
|
813
|
-
const
|
|
814
|
-
|
|
815
|
-
const rawContent = message?.content;
|
|
816
|
-
if (message === void 0 || message === null || typeof rawContent !== "string") {
|
|
820
|
+
const rawContent = paid.value.content;
|
|
821
|
+
if (typeof rawContent !== "string") {
|
|
817
822
|
throw new CaptureIntegrityError(
|
|
818
|
-
`chat response for scenario "${scenario.id}" turn ${i} is malformed: expected
|
|
823
|
+
`chat response for scenario "${scenario.id}" turn ${i} is malformed: expected content to be a string, got ${rawContent === null ? "null" : typeof rawContent}`
|
|
819
824
|
);
|
|
820
825
|
}
|
|
821
826
|
const content = rawContent;
|
|
@@ -903,8 +908,7 @@ async function executeScenario(tc, scenario, config) {
|
|
|
903
908
|
costLedger,
|
|
904
909
|
costPhase: config.costPhase ?? "benchmark.judge",
|
|
905
910
|
costTags,
|
|
906
|
-
signal: config.signal
|
|
907
|
-
tcloudMaximumAttempts: config.tcloudMaximumAttempts
|
|
911
|
+
signal: config.signal
|
|
908
912
|
};
|
|
909
913
|
const judgeResults = [];
|
|
910
914
|
let failedJudges = 0;
|
|
@@ -923,7 +927,7 @@ async function executeScenario(tc, scenario, config) {
|
|
|
923
927
|
console.log(` ${judgeName} retry ${attempt}/2 (waiting ${wait / 1e3}s)`);
|
|
924
928
|
await sleep2(wait);
|
|
925
929
|
}
|
|
926
|
-
const scores = await judge(
|
|
930
|
+
const scores = await judge(chat, judgeInput);
|
|
927
931
|
judgeResults.push(scores);
|
|
928
932
|
await sleep2(3e3);
|
|
929
933
|
lastError = void 0;
|
|
@@ -983,10 +987,10 @@ async function executeScenario(tc, scenario, config) {
|
|
|
983
987
|
|
|
984
988
|
// src/benchmark.ts
|
|
985
989
|
var BenchmarkRunner = class {
|
|
986
|
-
|
|
990
|
+
chat;
|
|
987
991
|
config;
|
|
988
|
-
constructor(
|
|
989
|
-
this.
|
|
992
|
+
constructor(chat, config) {
|
|
993
|
+
this.chat = chat;
|
|
990
994
|
this.config = config;
|
|
991
995
|
}
|
|
992
996
|
async run(scenarios) {
|
|
@@ -1008,13 +1012,12 @@ var BenchmarkRunner = class {
|
|
|
1008
1012
|
console.log(`[${i + 1}/${toRun.length}] ${scenario.id} (${scenario.persona})`);
|
|
1009
1013
|
console.log(` thesis: ${scenario.thesis}`);
|
|
1010
1014
|
console.log(` turns: ${scenario.turns.length}`);
|
|
1011
|
-
const result = await executeScenario(this.
|
|
1015
|
+
const result = await executeScenario(this.chat, scenario, {
|
|
1012
1016
|
systemPrompt: this.config.systemPrompt,
|
|
1013
1017
|
model: this.config.model,
|
|
1014
1018
|
judges: this.config.judges,
|
|
1015
1019
|
costLedger,
|
|
1016
|
-
costTags
|
|
1017
|
-
tcloudMaximumAttempts: this.config.tcloudMaximumAttempts
|
|
1020
|
+
costTags
|
|
1018
1021
|
});
|
|
1019
1022
|
results.push(result);
|
|
1020
1023
|
for (const turn of result.turns) {
|
|
@@ -1717,19 +1720,17 @@ var RIGOR_STANCE = {
|
|
|
1717
1720
|
relentless: "Your stance: a senior partner reviewing this work for a client who will litigate if it is wrong. You interrogate every claim. You accept nothing undefended. You find the single weakest point in every answer and attack it. Courteous, never satisfied."
|
|
1718
1721
|
};
|
|
1719
1722
|
var AgentDriver = class {
|
|
1720
|
-
|
|
1723
|
+
chat;
|
|
1721
1724
|
client;
|
|
1722
1725
|
driverModel;
|
|
1723
1726
|
productContext;
|
|
1724
1727
|
costLedger;
|
|
1725
|
-
|
|
1726
|
-
|
|
1727
|
-
this.tc = tc;
|
|
1728
|
+
constructor(chat, config) {
|
|
1729
|
+
this.chat = chat;
|
|
1728
1730
|
this.client = config.client;
|
|
1729
1731
|
this.driverModel = config.driverModel ?? "claude-sonnet-4-6";
|
|
1730
1732
|
this.productContext = config.productContext ?? "";
|
|
1731
1733
|
this.costLedger = config.costLedger ?? new CostLedger();
|
|
1732
|
-
this.tcloudMaximumAttempts = config.tcloudMaximumAttempts;
|
|
1733
1734
|
}
|
|
1734
1735
|
/**
|
|
1735
1736
|
* Run a persona through the product.
|
|
@@ -1812,15 +1813,14 @@ var AgentDriver = class {
|
|
|
1812
1813
|
}
|
|
1813
1814
|
/** Use the driver LLM to decide what the "user" says next */
|
|
1814
1815
|
async decideNextMessage(persona, state, history, costTags) {
|
|
1815
|
-
return decideNextUserTurn(this.
|
|
1816
|
+
return decideNextUserTurn(this.chat, {
|
|
1816
1817
|
persona,
|
|
1817
1818
|
state,
|
|
1818
1819
|
history,
|
|
1819
1820
|
productContext: this.productContext,
|
|
1820
1821
|
model: this.driverModel,
|
|
1821
1822
|
costLedger: this.costLedger,
|
|
1822
|
-
costTags
|
|
1823
|
-
tcloudMaximumAttempts: this.tcloudMaximumAttempts
|
|
1823
|
+
costTags
|
|
1824
1824
|
});
|
|
1825
1825
|
}
|
|
1826
1826
|
/** Handle pending approvals based on persona feedback patterns */
|
|
@@ -1925,7 +1925,7 @@ COMPLETION: drive toward the goal's real acceptance check. Do not declare done \
|
|
|
1925
1925
|
|
|
1926
1926
|
Output ONLY your next instruction to the worker \u2014 direct, detailed, actionable, in the first person as the driver. No meta-commentary, no preamble.`;
|
|
1927
1927
|
}
|
|
1928
|
-
async function decideNextUserTurn(
|
|
1928
|
+
async function decideNextUserTurn(chat, opts) {
|
|
1929
1929
|
const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
|
|
1930
1930
|
const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet \u2014 this is the first message)";
|
|
1931
1931
|
const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
|
|
@@ -1951,14 +1951,13 @@ ${lastResponse}` : "No conversation yet. Send your opening message \u2014 in cha
|
|
|
1951
1951
|
actor: "decideNextUserTurn",
|
|
1952
1952
|
model,
|
|
1953
1953
|
tags: opts.costTags,
|
|
1954
|
-
maximumCharge:
|
|
1955
|
-
execute: () =>
|
|
1956
|
-
receipt:
|
|
1954
|
+
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
|
|
1955
|
+
execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
|
|
1956
|
+
receipt: costReceiptFromLlm,
|
|
1957
|
+
receiptFromError: costReceiptFromLlmError
|
|
1957
1958
|
});
|
|
1958
1959
|
if (!paid.succeeded) throw paid.error;
|
|
1959
|
-
|
|
1960
|
-
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
1961
|
-
return content.trim();
|
|
1960
|
+
return paid.value.content.trim();
|
|
1962
1961
|
}
|
|
1963
1962
|
|
|
1964
1963
|
// src/feedback-trajectory.ts
|
|
@@ -4842,7 +4841,7 @@ function assertNonNegative(n, name) {
|
|
|
4842
4841
|
|
|
4843
4842
|
// src/eval-trace-store.ts
|
|
4844
4843
|
function runScore(record) {
|
|
4845
|
-
return
|
|
4844
|
+
return trainingScore(record);
|
|
4846
4845
|
}
|
|
4847
4846
|
function matches(record, f) {
|
|
4848
4847
|
if (f.experimentId && record.experimentId !== f.experimentId) return false;
|
|
@@ -4938,13 +4937,19 @@ var EvalTraceStore = class {
|
|
|
4938
4937
|
* Highest-scoring run for a scenario (optionally restricted to a candidate).
|
|
4939
4938
|
* Returns null when no run matches. Ties resolve to the earliest-appended run
|
|
4940
4939
|
* so the result is stable.
|
|
4940
|
+
*
|
|
4941
|
+
* Runs flagged as gamed are DROPPED, not zeroed. The caller's use for this is
|
|
4942
|
+
* few-shot seeding — the returned trajectory becomes an example to copy — so
|
|
4943
|
+
* the same rule as SFT applies: a faked success must not be in the candidate
|
|
4944
|
+
* set at all. When every run for the scenario is gated the honest answer is
|
|
4945
|
+
* `null` (no exemplar), never the least-bad fake.
|
|
4941
4946
|
*/
|
|
4942
4947
|
async getBest(scenarioId, opts = {}) {
|
|
4943
|
-
const rows = await this.query({
|
|
4948
|
+
const rows = (await this.query({
|
|
4944
4949
|
scenarioId,
|
|
4945
4950
|
candidateId: opts.candidateId,
|
|
4946
4951
|
splitTag: opts.splitTag
|
|
4947
|
-
});
|
|
4952
|
+
})).filter((r) => !isRealnessGated(r));
|
|
4948
4953
|
const scored = rows.flatMap((record) => {
|
|
4949
4954
|
const score = runScore(record);
|
|
4950
4955
|
return score === void 0 ? [] : [{ record, score }];
|
|
@@ -4966,6 +4971,9 @@ var EvalTraceStore = class {
|
|
|
4966
4971
|
* ran a scenario more than once, its best `runScore` for that scenario is
|
|
4967
4972
|
* used. Throws when there is no paired scenario — an unpaired "comparison" is
|
|
4968
4973
|
* not one.
|
|
4974
|
+
*
|
|
4975
|
+
* Realness-gated runs are excluded and counted in `realnessGatedRuns`, never
|
|
4976
|
+
* folded in as a zero.
|
|
4969
4977
|
*/
|
|
4970
4978
|
async compareRuns(candidateA, candidateB) {
|
|
4971
4979
|
if (candidateA === candidateB) {
|
|
@@ -4974,12 +4982,17 @@ var EvalTraceStore = class {
|
|
|
4974
4982
|
);
|
|
4975
4983
|
}
|
|
4976
4984
|
const rows = await this.backend.load();
|
|
4985
|
+
let realnessGatedRuns = 0;
|
|
4977
4986
|
const bestByScenario = (candidate) => {
|
|
4978
4987
|
const m = /* @__PURE__ */ new Map();
|
|
4979
4988
|
for (const r of rows) {
|
|
4980
4989
|
if (r.candidateId !== candidate) continue;
|
|
4981
4990
|
const sid = r.scenarioId;
|
|
4982
4991
|
if (!sid) continue;
|
|
4992
|
+
if (isRealnessGated(r)) {
|
|
4993
|
+
realnessGatedRuns++;
|
|
4994
|
+
continue;
|
|
4995
|
+
}
|
|
4983
4996
|
const s = runScore(r);
|
|
4984
4997
|
if (s === void 0) continue;
|
|
4985
4998
|
const prev = m.get(sid);
|
|
@@ -4992,7 +5005,7 @@ var EvalTraceStore = class {
|
|
|
4992
5005
|
const paired = [...aScores.keys()].filter((sid) => bScores.has(sid)).sort();
|
|
4993
5006
|
if (paired.length === 0) {
|
|
4994
5007
|
throw new ValidationError(
|
|
4995
|
-
`EvalTraceStore.compareRuns: "${candidateA}" and "${candidateB}" share no scenario (need scenarioId on records)`
|
|
5008
|
+
realnessGatedRuns > 0 ? `EvalTraceStore.compareRuns: "${candidateA}" and "${candidateB}" share no scenario with honest runs on both sides (${realnessGatedRuns} run(s) were realness-gated)` : `EvalTraceStore.compareRuns: "${candidateA}" and "${candidateB}" share no scenario (need scenarioId on records)`
|
|
4996
5009
|
);
|
|
4997
5010
|
}
|
|
4998
5011
|
let sumA = 0;
|
|
@@ -5020,7 +5033,8 @@ var EvalTraceStore = class {
|
|
|
5020
5033
|
meanDelta: meanB - meanA,
|
|
5021
5034
|
bWins,
|
|
5022
5035
|
ties,
|
|
5023
|
-
aWins
|
|
5036
|
+
aWins,
|
|
5037
|
+
realnessGatedRuns
|
|
5024
5038
|
};
|
|
5025
5039
|
}
|
|
5026
5040
|
};
|
|
@@ -9610,16 +9624,20 @@ var HeldOutGate = class {
|
|
|
9610
9624
|
const candidateId = inferCandidateId2(candidate, this.baselineKey);
|
|
9611
9625
|
const baselineId = this.baselineKey;
|
|
9612
9626
|
assertScenarioIdentities([...candidate, ...baseline]);
|
|
9613
|
-
const
|
|
9614
|
-
const
|
|
9615
|
-
const
|
|
9616
|
-
const
|
|
9627
|
+
const realnessGatedRuns = [...candidate, ...baseline].filter(isRealnessGated).length;
|
|
9628
|
+
const honestCandidate = candidate.filter((run) => !isRealnessGated(run));
|
|
9629
|
+
const honestBaseline = baseline.filter((run) => !isRealnessGated(run));
|
|
9630
|
+
const candidateSearch = scoredRuns(honestCandidate, "search");
|
|
9631
|
+
const baselineSearch = scoredRuns(honestBaseline, "search");
|
|
9632
|
+
const candidateHoldout = scoredRuns(honestCandidate, "holdout");
|
|
9633
|
+
const baselineHoldout = scoredRuns(honestBaseline, "holdout");
|
|
9617
9634
|
const searchPairing = pairRunRecords(baselineSearch, candidateSearch);
|
|
9618
9635
|
const holdoutPairing = pairRunRecords(baselineHoldout, candidateHoldout);
|
|
9619
|
-
const
|
|
9620
|
-
const
|
|
9621
|
-
const
|
|
9622
|
-
const
|
|
9636
|
+
const splitScoreOf = (run, split) => observedSplitScore(run, split);
|
|
9637
|
+
const beforeSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "search"));
|
|
9638
|
+
const afterSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "search"));
|
|
9639
|
+
const beforeHoldout = holdoutPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "holdout"));
|
|
9640
|
+
const afterHoldout = holdoutPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "holdout"));
|
|
9623
9641
|
const productiveRuns = beforeHoldout.length;
|
|
9624
9642
|
const candidateSearchMean = meanOrNull(afterSearch);
|
|
9625
9643
|
const candidateHoldoutMean = meanOrNull(afterHoldout);
|
|
@@ -9638,7 +9656,8 @@ var HeldOutGate = class {
|
|
|
9638
9656
|
overfitGap,
|
|
9639
9657
|
baselineOverfitGap,
|
|
9640
9658
|
medianCandidateCost,
|
|
9641
|
-
medianBaselineCost
|
|
9659
|
+
medianBaselineCost,
|
|
9660
|
+
realnessGatedRuns
|
|
9642
9661
|
};
|
|
9643
9662
|
const missingSplitScores = [
|
|
9644
9663
|
candidateSearch.length === 0 ? "candidate search" : null,
|
|
@@ -9755,10 +9774,12 @@ function assertScenarioIdentities(runs) {
|
|
|
9755
9774
|
}
|
|
9756
9775
|
}
|
|
9757
9776
|
}
|
|
9758
|
-
function scoredRuns(runs,
|
|
9759
|
-
return runs.filter(
|
|
9760
|
-
(run
|
|
9761
|
-
|
|
9777
|
+
function scoredRuns(runs, split) {
|
|
9778
|
+
return runs.filter((run) => {
|
|
9779
|
+
if (run.splitTag !== split) return false;
|
|
9780
|
+
const v = observedSplitScore(run, split);
|
|
9781
|
+
return typeof v === "number" && Number.isFinite(v);
|
|
9782
|
+
});
|
|
9762
9783
|
}
|
|
9763
9784
|
function meanOrNull(xs) {
|
|
9764
9785
|
if (xs.length === 0) return null;
|
|
@@ -10315,7 +10336,7 @@ async function tracedAnalyzeTraces(input, options, traceOpts) {
|
|
|
10315
10336
|
|
|
10316
10337
|
// src/traced-judges.ts
|
|
10317
10338
|
function traceJudge(judge, judgeName, opts) {
|
|
10318
|
-
return async (
|
|
10339
|
+
return async (chat, input) => {
|
|
10319
10340
|
const span = await opts.emitter.span({
|
|
10320
10341
|
kind: "llm",
|
|
10321
10342
|
name: `judge:${judgeName}`,
|
|
@@ -10326,7 +10347,7 @@ function traceJudge(judge, judgeName, opts) {
|
|
|
10326
10347
|
}
|
|
10327
10348
|
});
|
|
10328
10349
|
try {
|
|
10329
|
-
const scores = await judge(
|
|
10350
|
+
const scores = await judge(chat, input);
|
|
10330
10351
|
const composite = scores.length > 0 ? scores.reduce((sum4, s) => sum4 + s.score, 0) / scores.length : 0;
|
|
10331
10352
|
await span.end({
|
|
10332
10353
|
attributes: {
|
|
@@ -10344,7 +10365,7 @@ function traceJudge(judge, judgeName, opts) {
|
|
|
10344
10365
|
};
|
|
10345
10366
|
}
|
|
10346
10367
|
function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
10347
|
-
return async (
|
|
10368
|
+
return async (chat, input) => {
|
|
10348
10369
|
const ensembleSpan = await opts.emitter.span({
|
|
10349
10370
|
kind: "custom",
|
|
10350
10371
|
name: "judge:ensemble",
|
|
@@ -10365,7 +10386,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
|
10365
10386
|
parentSpanId: ensembleSpan.span.spanId
|
|
10366
10387
|
});
|
|
10367
10388
|
try {
|
|
10368
|
-
const scores = await tracedFn(
|
|
10389
|
+
const scores = await tracedFn(chat, input);
|
|
10369
10390
|
allScores.push(...scores);
|
|
10370
10391
|
} catch (err) {
|
|
10371
10392
|
if (!(err instanceof JudgeParseError)) throw err;
|
|
@@ -11387,6 +11408,7 @@ ${failures.join("\n")}`);
|
|
|
11387
11408
|
}
|
|
11388
11409
|
export {
|
|
11389
11410
|
AGENT_PROFILE_KINDS,
|
|
11411
|
+
ATIF_SCHEMA_VERSION,
|
|
11390
11412
|
ATTESTATION_ALGORITHM,
|
|
11391
11413
|
AgentDriver,
|
|
11392
11414
|
AgentEvalError,
|
|
@@ -11439,6 +11461,7 @@ export {
|
|
|
11439
11461
|
FileSystemRawProviderSink,
|
|
11440
11462
|
FileSystemTraceStore,
|
|
11441
11463
|
FindingsStore,
|
|
11464
|
+
HARBOR_IMPORT_GAP,
|
|
11442
11465
|
HARNESS_NATIVE_MODEL,
|
|
11443
11466
|
HeldOutGate,
|
|
11444
11467
|
HoldoutAuditor,
|
|
@@ -11532,7 +11555,6 @@ export {
|
|
|
11532
11555
|
ValidationError,
|
|
11533
11556
|
VerificationError,
|
|
11534
11557
|
acquisitionPlansForKnowledgeGaps,
|
|
11535
|
-
adversarialJudge,
|
|
11536
11558
|
agentProfileCellHashMaterial,
|
|
11537
11559
|
agentProfileCellKey,
|
|
11538
11560
|
agentProfileHash,
|
|
@@ -11558,6 +11580,8 @@ export {
|
|
|
11558
11580
|
assertCapabilityHeadroom,
|
|
11559
11581
|
assertCrossFamily,
|
|
11560
11582
|
assertLlmRoute,
|
|
11583
|
+
assertMinted,
|
|
11584
|
+
assertMintedLines,
|
|
11561
11585
|
assertModelsServed,
|
|
11562
11586
|
assertNoHiddenLeak,
|
|
11563
11587
|
assertProductBenchmarkRun,
|
|
@@ -11617,9 +11641,7 @@ export {
|
|
|
11617
11641
|
claudeCodeSupervisorRunReader,
|
|
11618
11642
|
cliffsDelta,
|
|
11619
11643
|
clusteredPairedBinary,
|
|
11620
|
-
codeExecutionJudge,
|
|
11621
11644
|
cohensD,
|
|
11622
|
-
coherenceJudge,
|
|
11623
11645
|
collectionPreserved,
|
|
11624
11646
|
commentsForSource,
|
|
11625
11647
|
commitBisect,
|
|
@@ -11654,9 +11676,7 @@ export {
|
|
|
11654
11676
|
createAnalystAi,
|
|
11655
11677
|
createAntiSlopJudge,
|
|
11656
11678
|
createChatClient,
|
|
11657
|
-
createCustomJudge,
|
|
11658
11679
|
createDefaultReviewer,
|
|
11659
|
-
createDomainExpertJudge,
|
|
11660
11680
|
createFeedbackTrajectory,
|
|
11661
11681
|
createIntentMatchJudge,
|
|
11662
11682
|
createLlmCorrectnessChecker,
|
|
@@ -11677,7 +11697,6 @@ export {
|
|
|
11677
11697
|
decideReferenceReplayRunPromotion,
|
|
11678
11698
|
defaultBlendWeights,
|
|
11679
11699
|
defaultIsMaterial,
|
|
11680
|
-
defaultJudges,
|
|
11681
11700
|
defaultProviderRedactor,
|
|
11682
11701
|
defaultReferenceReplayMatcher,
|
|
11683
11702
|
defaultTraceInsightPanel,
|
|
@@ -11738,6 +11757,7 @@ export {
|
|
|
11738
11757
|
formatDriverReport,
|
|
11739
11758
|
formatFindings,
|
|
11740
11759
|
formatScorecardDiff,
|
|
11760
|
+
fromHarborTrajectory,
|
|
11741
11761
|
gainHistogram,
|
|
11742
11762
|
gateTreatmentApplied,
|
|
11743
11763
|
gateTreatmentFromMetrics,
|
|
@@ -11777,6 +11797,7 @@ export {
|
|
|
11777
11797
|
isModelPriced,
|
|
11778
11798
|
isOtelConfigured,
|
|
11779
11799
|
isOtlpModelCall,
|
|
11800
|
+
isRealnessGated,
|
|
11780
11801
|
isRetrievalSpan,
|
|
11781
11802
|
isRolloutLine,
|
|
11782
11803
|
isRunRecord,
|
|
@@ -11829,6 +11850,8 @@ export {
|
|
|
11829
11850
|
notBlocked,
|
|
11830
11851
|
objectiveEval,
|
|
11831
11852
|
observeAll,
|
|
11853
|
+
observedScore,
|
|
11854
|
+
observedSplitScore,
|
|
11832
11855
|
otelRunCompleteHook,
|
|
11833
11856
|
otlpRowsToRunRecords,
|
|
11834
11857
|
otlpRowsToTraceRunRecords,
|
|
@@ -11891,6 +11914,7 @@ export {
|
|
|
11891
11914
|
referenceReplayScenarioToRunScore,
|
|
11892
11915
|
regexMatch,
|
|
11893
11916
|
regexMatches,
|
|
11917
|
+
relabelImportedSplit,
|
|
11894
11918
|
renderMarkdownReport,
|
|
11895
11919
|
renderPlaybookMarkdown,
|
|
11896
11920
|
renderPreferenceMemoryMarkdown,
|
|
@@ -11911,7 +11935,6 @@ export {
|
|
|
11911
11935
|
researchReport,
|
|
11912
11936
|
resolveModelPricing,
|
|
11913
11937
|
resolveSeat,
|
|
11914
|
-
rolloutReward,
|
|
11915
11938
|
rollupSupervisorRuns,
|
|
11916
11939
|
roundTripRunRecord,
|
|
11917
11940
|
routeFields,
|
|
@@ -11948,6 +11971,7 @@ export {
|
|
|
11948
11971
|
scoreContinuity,
|
|
11949
11972
|
scoreFromEvals,
|
|
11950
11973
|
scoreKnowledgeReadiness,
|
|
11974
|
+
scoreOrigin,
|
|
11951
11975
|
scorePrReviewComments,
|
|
11952
11976
|
scorePrReviewSource,
|
|
11953
11977
|
scoreRedTeamOutput,
|
|
@@ -11979,6 +12003,8 @@ export {
|
|
|
11979
12003
|
textInSnapshot,
|
|
11980
12004
|
throwIfRunIncomplete,
|
|
11981
12005
|
toAgentProfileJson,
|
|
12006
|
+
toHarborTrajectories,
|
|
12007
|
+
toHarborTrajectory,
|
|
11982
12008
|
toJsonl,
|
|
11983
12009
|
toLangfuseEnvelope,
|
|
11984
12010
|
toOpenAiTool,
|
|
@@ -11995,6 +12021,8 @@ export {
|
|
|
11995
12021
|
traceJudgeEnsemble,
|
|
11996
12022
|
traceSpanKindToOpenInferenceKind,
|
|
11997
12023
|
tracedAnalyzeTraces,
|
|
12024
|
+
trainingReward,
|
|
12025
|
+
trainingScore,
|
|
11998
12026
|
typoMutator,
|
|
11999
12027
|
urlContains,
|
|
12000
12028
|
userQuestionsForKnowledgeGaps,
|