@tangle-network/agent-eval 0.124.0 → 0.126.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -35
- package/README.md +270 -189
- package/dist/analyst/index.d.ts +15 -145
- package/dist/analyst/index.js +33 -47
- package/dist/analyst/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +45 -162
- package/dist/benchmarks/index.js +8 -9
- package/dist/campaign/index.d.ts +3655 -5365
- package/dist/campaign/index.js +21 -95
- package/dist/{chunk-R226UZOI.js → chunk-474LBSOX.js} +2 -2
- package/dist/{chunk-W5B3ZGP3.js → chunk-4B7ZZHPX.js} +8 -6
- package/dist/{chunk-W5B3ZGP3.js.map → chunk-4B7ZZHPX.js.map} +1 -1
- package/dist/{chunk-DT7OXY3C.js → chunk-CM4OILD2.js} +535 -846
- package/dist/chunk-CM4OILD2.js.map +1 -0
- package/dist/{chunk-HM6V7F3M.js → chunk-FO7HEH76.js} +3 -3
- package/dist/chunk-IILEIWGW.js +635 -0
- package/dist/chunk-IILEIWGW.js.map +1 -0
- package/dist/{chunk-EQUK3RFS.js → chunk-J5SQWP6Y.js} +8 -5
- package/dist/chunk-J5SQWP6Y.js.map +1 -0
- package/dist/chunk-KO2PZOGP.js +4637 -0
- package/dist/chunk-KO2PZOGP.js.map +1 -0
- package/dist/{chunk-4Y7AAATF.js → chunk-LKKT3IVV.js} +574 -81
- package/dist/chunk-LKKT3IVV.js.map +1 -0
- package/dist/chunk-M7AH34KV.js +155 -0
- package/dist/chunk-M7AH34KV.js.map +1 -0
- package/dist/chunk-NTOV7RU5.js +7152 -0
- package/dist/chunk-NTOV7RU5.js.map +1 -0
- package/dist/{chunk-QFQZ3U3X.js → chunk-OCFJACJU.js} +2 -2
- package/dist/{chunk-GID26AN4.js → chunk-P22LJ3Y2.js} +4 -6
- package/dist/{chunk-GID26AN4.js.map → chunk-P22LJ3Y2.js.map} +1 -1
- package/dist/{chunk-SJT4OBVL.js → chunk-SDPM6554.js} +3 -3
- package/dist/{chunk-D5JZ7UDZ.js → chunk-UCLVDLCH.js} +136 -50
- package/dist/chunk-UCLVDLCH.js.map +1 -0
- package/dist/chunk-UI4YMIN2.js +105 -0
- package/dist/chunk-UI4YMIN2.js.map +1 -0
- package/dist/chunk-VBQ3CRKH.js +291 -0
- package/dist/chunk-VBQ3CRKH.js.map +1 -0
- package/dist/{chunk-JKDNAOF5.js → chunk-W4L6C2XT.js} +2 -2
- package/dist/{chunk-GRCDRKII.js → chunk-WS3NZZQQ.js} +58 -20
- package/dist/chunk-WS3NZZQQ.js.map +1 -0
- package/dist/cli.js +3 -3
- package/dist/contract/index.d.ts +3221 -3094
- package/dist/contract/index.js +173 -42
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +2 -3
- package/dist/fuzz.d.ts +14 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +8 -1
- package/dist/index.d.ts +208 -690
- package/dist/index.js +185 -500
- package/dist/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/rl.d.ts +5 -100
- package/dist/rl.js +4 -5
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +9 -1
- package/dist/rollout/index.js +6 -6
- package/dist/{run-campaign-I3JXKVAK.js → run-campaign-LVFKZCEU.js} +3 -3
- package/dist/supervisor-run/index.d.ts +156 -4
- package/dist/supervisor-run/index.js +14 -2
- package/dist/traces.js +2 -3
- package/dist/wire/index.d.ts +14 -1
- package/dist/wire/index.js +3 -3
- package/docs/campaign-proposers.md +363 -168
- package/docs/design/loop-taxonomy.md +142 -190
- package/docs/design.md +1 -1
- package/docs/distributed-driver.md +8 -11
- package/docs/feature-guide.md +20 -19
- package/docs/knowledge-readiness.md +2 -5
- package/docs/multi-shot-optimization.md +35 -27
- package/docs/rollout.md +5 -5
- package/package.json +4 -4
- package/dist/chunk-4Y7AAATF.js.map +0 -1
- package/dist/chunk-5PVZVCZB.js +0 -9190
- package/dist/chunk-5PVZVCZB.js.map +0 -1
- package/dist/chunk-A6GT67HT.js +0 -550
- package/dist/chunk-A6GT67HT.js.map +0 -1
- package/dist/chunk-D5JZ7UDZ.js.map +0 -1
- package/dist/chunk-DT7OXY3C.js.map +0 -1
- package/dist/chunk-EQUK3RFS.js.map +0 -1
- package/dist/chunk-GC4ATIKK.js +0 -317
- package/dist/chunk-GC4ATIKK.js.map +0 -1
- package/dist/chunk-GRCDRKII.js.map +0 -1
- package/dist/chunk-LOW3U7JZ.js +0 -328
- package/dist/chunk-LOW3U7JZ.js.map +0 -1
- package/dist/chunk-MGGFVCJ7.js +0 -288
- package/dist/chunk-MGGFVCJ7.js.map +0 -1
- package/dist/chunk-PMITBABE.js +0 -3841
- package/dist/chunk-PMITBABE.js.map +0 -1
- package/dist/chunk-R7ZRE2KV.js +0 -138
- package/dist/chunk-R7ZRE2KV.js.map +0 -1
- /package/dist/{chunk-R226UZOI.js.map → chunk-474LBSOX.js.map} +0 -0
- /package/dist/{chunk-HM6V7F3M.js.map → chunk-FO7HEH76.js.map} +0 -0
- /package/dist/{chunk-QFQZ3U3X.js.map → chunk-OCFJACJU.js.map} +0 -0
- /package/dist/{chunk-SJT4OBVL.js.map → chunk-SDPM6554.js.map} +0 -0
- /package/dist/{chunk-JKDNAOF5.js.map → chunk-W4L6C2XT.js.map} +0 -0
- /package/dist/{run-campaign-I3JXKVAK.js.map → run-campaign-LVFKZCEU.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -9,24 +9,26 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-SDPM6554.js";
|
|
13
13
|
import {
|
|
14
14
|
mintRolloutRows,
|
|
15
15
|
rolloutReward
|
|
16
|
-
} from "./chunk-
|
|
16
|
+
} from "./chunk-M7AH34KV.js";
|
|
17
17
|
import {
|
|
18
18
|
SUPERVISOR_RUN_SCHEMA,
|
|
19
19
|
analyzeSupervisorRun,
|
|
20
20
|
analyzeSupervisorRunSources,
|
|
21
|
+
claudeCodeSupervisorRunReader,
|
|
21
22
|
isUnavailable,
|
|
23
|
+
readClaudeCodeSupervisorRun,
|
|
22
24
|
renderSupervisorRunHeadline,
|
|
23
25
|
renderSupervisorRunMarkdown,
|
|
24
26
|
rollupSupervisorRuns,
|
|
25
27
|
showMeasured,
|
|
26
28
|
supervisorRunRolloutLines,
|
|
27
29
|
writeSupervisorRunReport
|
|
28
|
-
} from "./chunk-
|
|
29
|
-
import "./chunk-
|
|
30
|
+
} from "./chunk-LKKT3IVV.js";
|
|
31
|
+
import "./chunk-VBQ3CRKH.js";
|
|
30
32
|
import {
|
|
31
33
|
toJsonl,
|
|
32
34
|
toRewardRows,
|
|
@@ -44,7 +46,7 @@ import {
|
|
|
44
46
|
BENCHMARK_SPLIT_SEED,
|
|
45
47
|
benchmarks_exports,
|
|
46
48
|
deterministicSplit
|
|
47
|
-
} from "./chunk-
|
|
49
|
+
} from "./chunk-W4L6C2XT.js";
|
|
48
50
|
import {
|
|
49
51
|
DEFAULT_RULES,
|
|
50
52
|
classifyFailure,
|
|
@@ -85,7 +87,7 @@ import {
|
|
|
85
87
|
pairArms,
|
|
86
88
|
parseCorrectnessResponse,
|
|
87
89
|
verifyCompletion
|
|
88
|
-
} from "./chunk-
|
|
90
|
+
} from "./chunk-KO2PZOGP.js";
|
|
89
91
|
import {
|
|
90
92
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
91
93
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -96,7 +98,6 @@ import {
|
|
|
96
98
|
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
97
99
|
adversarialJudge,
|
|
98
100
|
buildReflectionPrompt,
|
|
99
|
-
campaignMeanComposite,
|
|
100
101
|
codeExecutionJudge,
|
|
101
102
|
coherenceJudge,
|
|
102
103
|
costReceiptFromTCloud,
|
|
@@ -106,9 +107,7 @@ import {
|
|
|
106
107
|
crowdingDistance,
|
|
107
108
|
defaultJudges,
|
|
108
109
|
dominates,
|
|
109
|
-
gepaProposer,
|
|
110
110
|
hashScenarios,
|
|
111
|
-
heldOutGate,
|
|
112
111
|
llmJudge,
|
|
113
112
|
maximumChargeForTCloudRequest,
|
|
114
113
|
paretoFrontier,
|
|
@@ -117,13 +116,12 @@ import {
|
|
|
117
116
|
redTeamDataset,
|
|
118
117
|
redTeamReport,
|
|
119
118
|
runCanaries,
|
|
120
|
-
runImprovementLoop,
|
|
121
119
|
runReferenceEquivalenceJudge,
|
|
122
120
|
scalarScore,
|
|
123
121
|
scoreRedTeamOutput,
|
|
124
122
|
surfaceContentHash,
|
|
125
123
|
toolNamesForRun
|
|
126
|
-
} from "./chunk-
|
|
124
|
+
} from "./chunk-NTOV7RU5.js";
|
|
127
125
|
import {
|
|
128
126
|
BackendIntegrityError,
|
|
129
127
|
assertRealAgentReceipts,
|
|
@@ -135,7 +133,7 @@ import {
|
|
|
135
133
|
inMemoryVerdictCache,
|
|
136
134
|
summarizeAgentReceiptIntegrity,
|
|
137
135
|
summarizeBackendIntegrity
|
|
138
|
-
} from "./chunk-
|
|
136
|
+
} from "./chunk-UCLVDLCH.js";
|
|
139
137
|
import {
|
|
140
138
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
141
139
|
FindingsStore,
|
|
@@ -148,47 +146,32 @@ import {
|
|
|
148
146
|
defaultIsMaterial,
|
|
149
147
|
diffFindings,
|
|
150
148
|
runSemanticConceptJudge
|
|
151
|
-
} from "./chunk-
|
|
152
|
-
import {
|
|
153
|
-
buildDefaultAnalystRegistry,
|
|
154
|
-
computeTraceMetrics,
|
|
155
|
-
createChatClient
|
|
156
|
-
} from "./chunk-A6GT67HT.js";
|
|
157
|
-
import "./chunk-HHWE3POT.js";
|
|
149
|
+
} from "./chunk-4B7ZZHPX.js";
|
|
158
150
|
import {
|
|
159
151
|
AnalystRegistry,
|
|
160
|
-
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
161
152
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
162
153
|
FAILURE_MODE_KIND_SPEC,
|
|
163
154
|
IMPROVEMENT_KIND_SPEC,
|
|
164
155
|
KNOWLEDGE_GAP_KIND_SPEC,
|
|
165
156
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
166
|
-
|
|
167
|
-
POLICY_EDIT_AXES,
|
|
168
|
-
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
169
|
-
POLICY_EDIT_TARGET_SURFACES,
|
|
170
|
-
PolicyEditValidationError,
|
|
171
|
-
admitPolicyEdit,
|
|
172
|
-
aggregateRunScore,
|
|
173
|
-
applyPolicyEditToSurface,
|
|
174
|
-
clamp01,
|
|
157
|
+
buildDefaultAnalystRegistry,
|
|
175
158
|
computeFindingId,
|
|
176
|
-
|
|
159
|
+
computeTraceMetrics,
|
|
177
160
|
createAnalystAi,
|
|
161
|
+
createChatClient,
|
|
178
162
|
createTraceAnalystKind,
|
|
179
|
-
isPolicyEdit,
|
|
180
163
|
makeFinding,
|
|
181
|
-
makePolicyEdit,
|
|
182
|
-
makePolicyEditCandidateRecord,
|
|
183
|
-
mapConcurrent,
|
|
184
|
-
policyEditFromFinding,
|
|
185
|
-
policyEditsFromFindings,
|
|
186
164
|
renderPriorFindings,
|
|
187
|
-
renderUpstreamFindings
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
165
|
+
renderUpstreamFindings
|
|
166
|
+
} from "./chunk-CM4OILD2.js";
|
|
167
|
+
import "./chunk-HHWE3POT.js";
|
|
168
|
+
import {
|
|
169
|
+
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
170
|
+
Mutex,
|
|
171
|
+
aggregateRunScore,
|
|
172
|
+
clamp01,
|
|
173
|
+
mapConcurrent
|
|
174
|
+
} from "./chunk-UI4YMIN2.js";
|
|
192
175
|
import {
|
|
193
176
|
allCriticalPassed,
|
|
194
177
|
controlFailureClassFromVerification,
|
|
@@ -209,7 +192,7 @@ import {
|
|
|
209
192
|
stopOnNoProgress,
|
|
210
193
|
stopOnRepeatedAction,
|
|
211
194
|
subjectiveEval
|
|
212
|
-
} from "./chunk-
|
|
195
|
+
} from "./chunk-474LBSOX.js";
|
|
213
196
|
import {
|
|
214
197
|
assertReleaseConfidence,
|
|
215
198
|
bootstrapCi,
|
|
@@ -219,7 +202,7 @@ import {
|
|
|
219
202
|
} from "./chunk-MOXWMGPC.js";
|
|
220
203
|
import {
|
|
221
204
|
runEvalCampaign
|
|
222
|
-
} from "./chunk-
|
|
205
|
+
} from "./chunk-P22LJ3Y2.js";
|
|
223
206
|
import "./chunk-ARU2PZFM.js";
|
|
224
207
|
import {
|
|
225
208
|
LlmCallError,
|
|
@@ -236,7 +219,7 @@ import {
|
|
|
236
219
|
maximumChargeForLlmRequest,
|
|
237
220
|
probeLlm,
|
|
238
221
|
stripFencedJson
|
|
239
|
-
} from "./chunk-
|
|
222
|
+
} from "./chunk-J5SQWP6Y.js";
|
|
240
223
|
import {
|
|
241
224
|
evaluateInterimReleaseConfidence,
|
|
242
225
|
pairedEvalueSequence
|
|
@@ -299,7 +282,7 @@ import {
|
|
|
299
282
|
costForTokenPricing,
|
|
300
283
|
costForUsage,
|
|
301
284
|
modelPriceKey
|
|
302
|
-
} from "./chunk-
|
|
285
|
+
} from "./chunk-WS3NZZQQ.js";
|
|
303
286
|
import {
|
|
304
287
|
MODEL_PRICING,
|
|
305
288
|
MetricsCollector,
|
|
@@ -338,7 +321,7 @@ import {
|
|
|
338
321
|
scoreTraceInsightReadiness,
|
|
339
322
|
tokenizeDomainWords,
|
|
340
323
|
traceAnalystOnRunComplete
|
|
341
|
-
} from "./chunk-
|
|
324
|
+
} from "./chunk-OCFJACJU.js";
|
|
342
325
|
import "./chunk-H5UD2323.js";
|
|
343
326
|
import {
|
|
344
327
|
extractUsage,
|
|
@@ -403,14 +386,26 @@ import {
|
|
|
403
386
|
llmSpanFromProvider
|
|
404
387
|
} from "./chunk-VQMK5FMP.js";
|
|
405
388
|
import {
|
|
389
|
+
AGENT_PROFILE_KINDS,
|
|
390
|
+
AgentProfileCellValidationError,
|
|
406
391
|
RunRecordValidationError,
|
|
392
|
+
agentProfileCellHashMaterial,
|
|
393
|
+
agentProfileCellKey,
|
|
394
|
+
assertRunAgentProfileCell,
|
|
395
|
+
buildAgentInterfaceProfileCell,
|
|
396
|
+
buildAgentProfileCell,
|
|
397
|
+
groupRunsByAgentProfileCell,
|
|
407
398
|
isRunRecord,
|
|
408
399
|
modelHasSnapshot,
|
|
409
400
|
parseRunRecordSafe,
|
|
401
|
+
requireAgentProfileCell,
|
|
410
402
|
resolveRunCostProvenance,
|
|
411
403
|
roundTripRunRecord,
|
|
412
|
-
|
|
413
|
-
|
|
404
|
+
toAgentProfileJson,
|
|
405
|
+
validateAgentProfileCell,
|
|
406
|
+
validateRunRecord,
|
|
407
|
+
verifyAgentProfileCell
|
|
408
|
+
} from "./chunk-IILEIWGW.js";
|
|
414
409
|
import {
|
|
415
410
|
FAILURE_CLASSES,
|
|
416
411
|
TRACE_SCHEMA_VERSION,
|
|
@@ -420,20 +415,6 @@ import {
|
|
|
420
415
|
isSandboxSpan,
|
|
421
416
|
isToolSpan
|
|
422
417
|
} from "./chunk-MA6HLL3S.js";
|
|
423
|
-
import {
|
|
424
|
-
AGENT_PROFILE_KINDS,
|
|
425
|
-
AgentProfileCellValidationError,
|
|
426
|
-
agentProfileCellHashMaterial,
|
|
427
|
-
agentProfileCellKey,
|
|
428
|
-
assertRunAgentProfileCell,
|
|
429
|
-
buildAgentInterfaceProfileCell,
|
|
430
|
-
buildAgentProfileCell,
|
|
431
|
-
groupRunsByAgentProfileCell,
|
|
432
|
-
requireAgentProfileCell,
|
|
433
|
-
toAgentProfileJson,
|
|
434
|
-
validateAgentProfileCell,
|
|
435
|
-
verifyAgentProfileCell
|
|
436
|
-
} from "./chunk-GC4ATIKK.js";
|
|
437
418
|
import {
|
|
438
419
|
canonicalize,
|
|
439
420
|
evaluateHypothesis,
|
|
@@ -10064,414 +10045,6 @@ function isOtelConfigured() {
|
|
|
10064
10045
|
return !!(typeof process !== "undefined" && process.env.OTEL_EXPORTER_OTLP_ENDPOINT);
|
|
10065
10046
|
}
|
|
10066
10047
|
|
|
10067
|
-
// src/traced-analyst.ts
|
|
10068
|
-
async function tracedAnalyzeTraces(input, options, traceOpts) {
|
|
10069
|
-
const parentSpan = await traceOpts.emitter.span({
|
|
10070
|
-
kind: "custom",
|
|
10071
|
-
name: "analyst:analyze-traces",
|
|
10072
|
-
parentSpanId: traceOpts.parentSpanId,
|
|
10073
|
-
attributes: {
|
|
10074
|
-
"analyst.question_length": input.question.length,
|
|
10075
|
-
"analyst.max_turns": options.maxTurns ?? 12,
|
|
10076
|
-
"analyst.max_subqueries": options.maxSubqueries ?? 4,
|
|
10077
|
-
"eval.phase": "analyst"
|
|
10078
|
-
}
|
|
10079
|
-
});
|
|
10080
|
-
const originalOnTurn = options.onTurn;
|
|
10081
|
-
const wrappedOptions = {
|
|
10082
|
-
...options,
|
|
10083
|
-
onTurn: async (turn) => {
|
|
10084
|
-
const turnSpan = await traceOpts.emitter.span({
|
|
10085
|
-
kind: "custom",
|
|
10086
|
-
name: `analyst:turn-${turn.turn}`,
|
|
10087
|
-
parentSpanId: parentSpan.span.spanId,
|
|
10088
|
-
attributes: {
|
|
10089
|
-
"analyst.stage": turn.stage,
|
|
10090
|
-
"analyst.turn": turn.turn,
|
|
10091
|
-
"analyst.is_error": turn.isError,
|
|
10092
|
-
"analyst.code_length": turn.code.length,
|
|
10093
|
-
"analyst.output_length": turn.output.length,
|
|
10094
|
-
"eval.phase": "analyst"
|
|
10095
|
-
}
|
|
10096
|
-
});
|
|
10097
|
-
if (turn.isError) {
|
|
10098
|
-
await turnSpan.fail("Turn produced an error");
|
|
10099
|
-
} else {
|
|
10100
|
-
await turnSpan.end();
|
|
10101
|
-
}
|
|
10102
|
-
if (originalOnTurn) await originalOnTurn(turn);
|
|
10103
|
-
}
|
|
10104
|
-
};
|
|
10105
|
-
try {
|
|
10106
|
-
const result = await analyzeTraces(input, wrappedOptions);
|
|
10107
|
-
await parentSpan.end({
|
|
10108
|
-
attributes: {
|
|
10109
|
-
"analyst.question_length": input.question.length,
|
|
10110
|
-
"analyst.turn_count": result.turnCount,
|
|
10111
|
-
"analyst.finding_count": result.findings.length,
|
|
10112
|
-
"analyst.answer_length": result.answer.length,
|
|
10113
|
-
"eval.phase": "analyst"
|
|
10114
|
-
}
|
|
10115
|
-
});
|
|
10116
|
-
return result;
|
|
10117
|
-
} catch (err) {
|
|
10118
|
-
await parentSpan.fail(err instanceof Error ? err : String(err));
|
|
10119
|
-
throw err;
|
|
10120
|
-
}
|
|
10121
|
-
}
|
|
10122
|
-
|
|
10123
|
-
// src/traced-judges.ts
|
|
10124
|
-
function traceJudge(judge, judgeName, opts) {
|
|
10125
|
-
return async (tc, input) => {
|
|
10126
|
-
const span = await opts.emitter.span({
|
|
10127
|
-
kind: "llm",
|
|
10128
|
-
name: `judge:${judgeName}`,
|
|
10129
|
-
parentSpanId: opts.parentSpanId,
|
|
10130
|
-
attributes: {
|
|
10131
|
-
"judge.name": judgeName,
|
|
10132
|
-
"eval.phase": "judge"
|
|
10133
|
-
}
|
|
10134
|
-
});
|
|
10135
|
-
try {
|
|
10136
|
-
const scores2 = await judge(tc, input);
|
|
10137
|
-
const composite = scores2.length > 0 ? scores2.reduce((sum4, s) => sum4 + s.score, 0) / scores2.length : 0;
|
|
10138
|
-
await span.end({
|
|
10139
|
-
attributes: {
|
|
10140
|
-
"judge.name": judgeName,
|
|
10141
|
-
"judge.composite_score": composite,
|
|
10142
|
-
"judge.dimension_count": scores2.length,
|
|
10143
|
-
"eval.phase": "judge"
|
|
10144
|
-
}
|
|
10145
|
-
});
|
|
10146
|
-
return scores2;
|
|
10147
|
-
} catch (err) {
|
|
10148
|
-
await span.fail(err instanceof Error ? err : String(err));
|
|
10149
|
-
throw err;
|
|
10150
|
-
}
|
|
10151
|
-
};
|
|
10152
|
-
}
|
|
10153
|
-
function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
10154
|
-
return async (tc, input) => {
|
|
10155
|
-
const ensembleSpan = await opts.emitter.span({
|
|
10156
|
-
kind: "custom",
|
|
10157
|
-
name: "judge:ensemble",
|
|
10158
|
-
parentSpanId: opts.parentSpanId,
|
|
10159
|
-
attributes: {
|
|
10160
|
-
"judge.ensemble_size": judges.length,
|
|
10161
|
-
"eval.phase": "judge"
|
|
10162
|
-
}
|
|
10163
|
-
});
|
|
10164
|
-
try {
|
|
10165
|
-
const allScores = [];
|
|
10166
|
-
let failedJudges = 0;
|
|
10167
|
-
for (let i = 0; i < judges.length; i++) {
|
|
10168
|
-
const judge = judges[i];
|
|
10169
|
-
const name = judgeNames[i] ?? `judge_${i}`;
|
|
10170
|
-
const tracedFn = traceJudge(judge, name, {
|
|
10171
|
-
emitter: opts.emitter,
|
|
10172
|
-
parentSpanId: ensembleSpan.span.spanId
|
|
10173
|
-
});
|
|
10174
|
-
try {
|
|
10175
|
-
const scores2 = await tracedFn(tc, input);
|
|
10176
|
-
allScores.push(...scores2);
|
|
10177
|
-
} catch (err) {
|
|
10178
|
-
if (!(err instanceof JudgeParseError)) throw err;
|
|
10179
|
-
failedJudges++;
|
|
10180
|
-
}
|
|
10181
|
-
}
|
|
10182
|
-
const composite = allScores.length > 0 ? allScores.reduce((sum4, s) => sum4 + s.score, 0) / allScores.length : 0;
|
|
10183
|
-
await ensembleSpan.end({
|
|
10184
|
-
attributes: {
|
|
10185
|
-
"judge.ensemble_size": judges.length,
|
|
10186
|
-
"judge.composite_score": composite,
|
|
10187
|
-
"judge.total_dimensions": allScores.length,
|
|
10188
|
-
"judge.failed_judges": failedJudges,
|
|
10189
|
-
"eval.phase": "judge"
|
|
10190
|
-
}
|
|
10191
|
-
});
|
|
10192
|
-
return allScores;
|
|
10193
|
-
} catch (err) {
|
|
10194
|
-
await ensembleSpan.fail(err instanceof Error ? err : String(err));
|
|
10195
|
-
throw err;
|
|
10196
|
-
}
|
|
10197
|
-
};
|
|
10198
|
-
}
|
|
10199
|
-
|
|
10200
|
-
// src/campaign/distillation/agreement-judge.ts
|
|
10201
|
-
var AGREEMENT_DIM = "agreement";
|
|
10202
|
-
function buildAgreementJudge(options) {
|
|
10203
|
-
const name = options.name ?? "gold-agreement";
|
|
10204
|
-
const goldOnly = options.goldOnly ?? true;
|
|
10205
|
-
const declaredDims = options.dimensionKeys ?? [AGREEMENT_DIM];
|
|
10206
|
-
return {
|
|
10207
|
-
name,
|
|
10208
|
-
dimensions: declaredDims.map((key) => ({
|
|
10209
|
-
key,
|
|
10210
|
-
description: `Per-field agreement between the produced label and the gold label on '${key}'`
|
|
10211
|
-
})),
|
|
10212
|
-
appliesTo: goldOnly ? (scenario) => scenario.kind === "gold" : void 0,
|
|
10213
|
-
score({ artifact, scenario }) {
|
|
10214
|
-
const { score, dimensions } = options.compareLabels(artifact, scenario.label);
|
|
10215
|
-
if (!Number.isFinite(score) || score < 0 || score > 1) {
|
|
10216
|
-
throw new Error(
|
|
10217
|
-
`buildAgreementJudge: comparator returned out-of-range score ${score} for scenario '${scenario.id}' (must be in [0,1])`
|
|
10218
|
-
);
|
|
10219
|
-
}
|
|
10220
|
-
const outDims = { [AGREEMENT_DIM]: score, ...dimensions };
|
|
10221
|
-
const weakest = Object.entries(dimensions).sort((a, b) => a[1] - b[1])[0];
|
|
10222
|
-
const notes = weakest ? `agreement ${score.toFixed(3)}; weakest field '${weakest[0]}' (${weakest[1].toFixed(3)})` : `agreement ${score.toFixed(3)}`;
|
|
10223
|
-
return { composite: score, dimensions: outDims, notes };
|
|
10224
|
-
}
|
|
10225
|
-
};
|
|
10226
|
-
}
|
|
10227
|
-
function fieldAgreement(spec) {
|
|
10228
|
-
const categorical = spec.categorical ?? [];
|
|
10229
|
-
const array = spec.array ?? [];
|
|
10230
|
-
if (categorical.length === 0 && array.length === 0) {
|
|
10231
|
-
throw new Error("fieldAgreement: at least one categorical or array field is required");
|
|
10232
|
-
}
|
|
10233
|
-
return (produced, gold) => {
|
|
10234
|
-
const p = produced ?? {};
|
|
10235
|
-
const g = gold ?? {};
|
|
10236
|
-
const dimensions = {};
|
|
10237
|
-
for (const field of categorical) {
|
|
10238
|
-
dimensions[field] = categoricalAgreement(p[field], g[field]);
|
|
10239
|
-
}
|
|
10240
|
-
for (const field of array) {
|
|
10241
|
-
dimensions[field] = jaccard(asArray(p[field]), asArray(g[field]));
|
|
10242
|
-
}
|
|
10243
|
-
const values = Object.values(dimensions);
|
|
10244
|
-
const score = values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length;
|
|
10245
|
-
return { score, dimensions };
|
|
10246
|
-
};
|
|
10247
|
-
}
|
|
10248
|
-
function categoricalAgreement(produced, gold) {
|
|
10249
|
-
if (produced === void 0 && gold === void 0) return 1;
|
|
10250
|
-
return normalizeScalar(produced) === normalizeScalar(gold) ? 1 : 0;
|
|
10251
|
-
}
|
|
10252
|
-
function normalizeScalar(value) {
|
|
10253
|
-
if (value === void 0) return "__undefined__";
|
|
10254
|
-
if (value === null) return "__null__";
|
|
10255
|
-
return JSON.stringify(value);
|
|
10256
|
-
}
|
|
10257
|
-
function asArray(value) {
|
|
10258
|
-
if (Array.isArray(value)) return value;
|
|
10259
|
-
if (value === void 0 || value === null) return [];
|
|
10260
|
-
return [value];
|
|
10261
|
-
}
|
|
10262
|
-
function jaccard(a, b) {
|
|
10263
|
-
const sa = new Set(a.map((x) => JSON.stringify(x)));
|
|
10264
|
-
const sb = new Set(b.map((x) => JSON.stringify(x)));
|
|
10265
|
-
if (sa.size === 0 && sb.size === 0) return 1;
|
|
10266
|
-
let inter = 0;
|
|
10267
|
-
for (const x of sa) if (sb.has(x)) inter++;
|
|
10268
|
-
const union = sa.size + sb.size - inter;
|
|
10269
|
-
return union === 0 ? 1 : inter / union;
|
|
10270
|
-
}
|
|
10271
|
-
|
|
10272
|
-
// src/campaign/distillation/gold-scenarios.ts
|
|
10273
|
-
import { readFileSync as readFileSync5 } from "fs";
|
|
10274
|
-
function loadGoldScenarios(jsonlPath) {
|
|
10275
|
-
const text = readFileSync5(jsonlPath, "utf8");
|
|
10276
|
-
return parseGoldJsonl(text, jsonlPath);
|
|
10277
|
-
}
|
|
10278
|
-
function parseGoldJsonl(text, sourceLabel = "<inline>") {
|
|
10279
|
-
const out = [];
|
|
10280
|
-
const lines = text.split("\n");
|
|
10281
|
-
for (let i = 0; i < lines.length; i++) {
|
|
10282
|
-
const raw = lines[i].trim();
|
|
10283
|
-
if (raw.length === 0) continue;
|
|
10284
|
-
let parsed;
|
|
10285
|
-
try {
|
|
10286
|
-
parsed = JSON.parse(raw);
|
|
10287
|
-
} catch (err) {
|
|
10288
|
-
throw new Error(
|
|
10289
|
-
`loadGoldScenarios: ${sourceLabel}:${i + 1} is not valid JSON \u2014 ${err instanceof Error ? err.message : String(err)}`
|
|
10290
|
-
);
|
|
10291
|
-
}
|
|
10292
|
-
const rawId = parsed.scenarioId ?? parsed.id;
|
|
10293
|
-
if (typeof rawId !== "string" || rawId.length === 0) {
|
|
10294
|
-
throw new Error(
|
|
10295
|
-
`loadGoldScenarios: ${sourceLabel}:${i + 1} missing string \`scenarioId\`/\`id\``
|
|
10296
|
-
);
|
|
10297
|
-
}
|
|
10298
|
-
const id = rawId.replace(/:/g, "__");
|
|
10299
|
-
if (parsed.input === void 0) {
|
|
10300
|
-
throw new Error(`loadGoldScenarios: ${sourceLabel}:${i + 1} (${rawId}) missing \`input\``);
|
|
10301
|
-
}
|
|
10302
|
-
if (parsed.label === void 0) {
|
|
10303
|
-
throw new Error(`loadGoldScenarios: ${sourceLabel}:${i + 1} (${rawId}) missing \`label\``);
|
|
10304
|
-
}
|
|
10305
|
-
const scenario = {
|
|
10306
|
-
id,
|
|
10307
|
-
kind: "gold",
|
|
10308
|
-
input: parsed.input,
|
|
10309
|
-
label: parsed.label
|
|
10310
|
-
};
|
|
10311
|
-
const tags = [];
|
|
10312
|
-
if (id !== rawId) tags.push(`gold-id:${rawId}`);
|
|
10313
|
-
if (parsed.split !== void 0) tags.push(`split:${parsed.split}`);
|
|
10314
|
-
if (tags.length > 0) scenario.tags = tags;
|
|
10315
|
-
out.push(scenario);
|
|
10316
|
-
}
|
|
10317
|
-
if (out.length === 0) {
|
|
10318
|
-
throw new Error(`loadGoldScenarios: ${sourceLabel} contained no gold records`);
|
|
10319
|
-
}
|
|
10320
|
-
return out;
|
|
10321
|
-
}
|
|
10322
|
-
function splitGold(scenarios, options = {}) {
|
|
10323
|
-
const testEveryNth = options.testEveryNth ?? 4;
|
|
10324
|
-
if (!Number.isInteger(testEveryNth) || testEveryNth < 2) {
|
|
10325
|
-
throw new Error("splitGold: testEveryNth must be an integer \u2265 2 (else train or test is empty)");
|
|
10326
|
-
}
|
|
10327
|
-
const train = [];
|
|
10328
|
-
const test = [];
|
|
10329
|
-
let implicitIndex = 0;
|
|
10330
|
-
for (const scenario of scenarios) {
|
|
10331
|
-
const explicit = explicitSplit(scenario);
|
|
10332
|
-
if (explicit === "train") {
|
|
10333
|
-
train.push(scenario);
|
|
10334
|
-
} else if (explicit === "test") {
|
|
10335
|
-
test.push(scenario);
|
|
10336
|
-
} else {
|
|
10337
|
-
if (implicitIndex % testEveryNth === 0) test.push(scenario);
|
|
10338
|
-
else train.push(scenario);
|
|
10339
|
-
implicitIndex += 1;
|
|
10340
|
-
}
|
|
10341
|
-
}
|
|
10342
|
-
return { train, test };
|
|
10343
|
-
}
|
|
10344
|
-
function explicitSplit(scenario) {
|
|
10345
|
-
for (const tag of scenario.tags ?? []) {
|
|
10346
|
-
if (tag === "split:train") return "train";
|
|
10347
|
-
if (tag === "split:test") return "test";
|
|
10348
|
-
}
|
|
10349
|
-
return void 0;
|
|
10350
|
-
}
|
|
10351
|
-
|
|
10352
|
-
// src/campaign/distillation/run-distillation.ts
|
|
10353
|
-
async function runDistillation(opts) {
|
|
10354
|
-
if (opts.train.length === 0) throw new Error("runDistillation: train split is empty");
|
|
10355
|
-
if (opts.holdout.length === 0) throw new Error("runDistillation: holdout split is empty");
|
|
10356
|
-
const chat = createChatClient(opts.llm);
|
|
10357
|
-
const render = opts.renderStudentPrompt ?? defaultRenderStudentPrompt;
|
|
10358
|
-
const parse = opts.parseStudentLabel ?? defaultParseStudentLabel;
|
|
10359
|
-
const runDir = opts.runDir ?? `.evolve/distillation/${Date.now()}`;
|
|
10360
|
-
const studentTemperature = opts.studentTemperature ?? 0;
|
|
10361
|
-
const studentMaxTokens = opts.studentMaxTokens ?? 1024;
|
|
10362
|
-
const proposer = gepaProposer({
|
|
10363
|
-
llm: opts.reflectionLlm,
|
|
10364
|
-
model: opts.optimizerModel,
|
|
10365
|
-
target: "a cheap single-shot analyst system prompt that reproduces an expensive workflow gold verdict",
|
|
10366
|
-
mutationPrimitives: opts.mutationPrimitives ?? DEFAULT_MUTATION_PRIMITIVES2,
|
|
10367
|
-
constraints: opts.constraints
|
|
10368
|
-
});
|
|
10369
|
-
const gate = opts.gate ?? heldOutGate({
|
|
10370
|
-
scenarios: opts.holdout,
|
|
10371
|
-
deltaThreshold: opts.deltaThreshold ?? 0
|
|
10372
|
-
});
|
|
10373
|
-
const loop = await runImprovementLoop({
|
|
10374
|
-
baselineSurface: opts.baselinePrompt,
|
|
10375
|
-
scenarios: opts.train,
|
|
10376
|
-
holdoutScenarios: opts.holdout,
|
|
10377
|
-
judges: [opts.judge],
|
|
10378
|
-
proposer,
|
|
10379
|
-
gate,
|
|
10380
|
-
autoOnPromote: "none",
|
|
10381
|
-
// the loop NEVER opens a PR — the caller decides
|
|
10382
|
-
populationSize: opts.populationSize ?? 4,
|
|
10383
|
-
maxGenerations: opts.maxGenerations ?? 3,
|
|
10384
|
-
reps: opts.reps ?? 1,
|
|
10385
|
-
runDir,
|
|
10386
|
-
// The student spends tokens; tracing must stay on (the proposer is wired and
|
|
10387
|
-
// runImprovementLoop refuses tracing='off' with a proposer).
|
|
10388
|
-
tracing: "on",
|
|
10389
|
-
dispatchWithSurface: async (surface, scenario, ctx) => {
|
|
10390
|
-
const prompt = render({
|
|
10391
|
-
surface: typeof surface === "string" ? surface : JSON.stringify(surface),
|
|
10392
|
-
input: scenario.input,
|
|
10393
|
-
scenarioId: scenario.id
|
|
10394
|
-
});
|
|
10395
|
-
const request = {
|
|
10396
|
-
model: opts.studentModel,
|
|
10397
|
-
messages: prompt,
|
|
10398
|
-
jsonMode: true,
|
|
10399
|
-
temperature: studentTemperature,
|
|
10400
|
-
maxTokens: studentMaxTokens
|
|
10401
|
-
};
|
|
10402
|
-
const paid = await ctx.cost.runPaidCall({
|
|
10403
|
-
actor: "distillation-student",
|
|
10404
|
-
model: opts.studentModel,
|
|
10405
|
-
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maxRetries: chat.maximumAttempts }),
|
|
10406
|
-
execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
|
|
10407
|
-
receipt: costReceiptFromLlm,
|
|
10408
|
-
receiptFromError: costReceiptFromLlmError
|
|
10409
|
-
});
|
|
10410
|
-
if (!paid.succeeded) throw paid.error;
|
|
10411
|
-
return parse(paid.value.content, scenario.id);
|
|
10412
|
-
}
|
|
10413
|
-
});
|
|
10414
|
-
const winnerPrompt = typeof loop.winnerSurface === "string" ? loop.winnerSurface : opts.baselinePrompt;
|
|
10415
|
-
const baseline = campaignMeanComposite(loop.baselineOnHoldout);
|
|
10416
|
-
const winner = campaignMeanComposite(loop.winnerOnHoldout);
|
|
10417
|
-
return {
|
|
10418
|
-
...loop,
|
|
10419
|
-
winnerPrompt,
|
|
10420
|
-
holdoutAgreement: { baseline, winner, delta: winner - baseline }
|
|
10421
|
-
};
|
|
10422
|
-
}
|
|
10423
|
-
var DEFAULT_MUTATION_PRIMITIVES2 = [
|
|
10424
|
-
"Add an explicit output-schema instruction so the model emits exactly the gold label fields as JSON.",
|
|
10425
|
-
"Add a one-line decision rule for each verdict field the student keeps getting wrong.",
|
|
10426
|
-
"Add a worked example mapping a representative input to its correct gold label.",
|
|
10427
|
-
"Tighten ambiguous phrasing that lets the student hedge instead of committing to a verdict.",
|
|
10428
|
-
"Add a guardrail that forces the student to set boolean risk flags (e.g. leak risk) when the triggering condition is present."
|
|
10429
|
-
];
|
|
10430
|
-
function defaultRenderStudentPrompt(args) {
|
|
10431
|
-
return [
|
|
10432
|
-
{ role: "system", content: args.surface },
|
|
10433
|
-
{
|
|
10434
|
-
role: "user",
|
|
10435
|
-
content: `Input:
|
|
10436
|
-
${stableStringify(args.input)}
|
|
10437
|
-
|
|
10438
|
-
Respond with ONLY a single JSON object \u2014 the verdict. No prose, no code fences.`
|
|
10439
|
-
}
|
|
10440
|
-
];
|
|
10441
|
-
}
|
|
10442
|
-
function defaultParseStudentLabel(rawContent, scenarioId) {
|
|
10443
|
-
const stripped = stripFence(rawContent).trim();
|
|
10444
|
-
if (stripped.length === 0) {
|
|
10445
|
-
throw new Error(`distillation student returned empty output for scenario '${scenarioId}'`);
|
|
10446
|
-
}
|
|
10447
|
-
try {
|
|
10448
|
-
return JSON.parse(stripped);
|
|
10449
|
-
} catch (err) {
|
|
10450
|
-
throw new Error(
|
|
10451
|
-
`distillation student returned non-JSON for scenario '${scenarioId}': ${err instanceof Error ? err.message : String(err)} \u2014 raw: ${stripped.slice(0, 200)}`
|
|
10452
|
-
);
|
|
10453
|
-
}
|
|
10454
|
-
}
|
|
10455
|
-
function stripFence(text) {
|
|
10456
|
-
const fenced = /```(?:json)?\s*([\s\S]*?)\s*```/.exec(text);
|
|
10457
|
-
return fenced ? fenced[1] ?? text : text;
|
|
10458
|
-
}
|
|
10459
|
-
function stableStringify(value) {
|
|
10460
|
-
return JSON.stringify(value, replacerSortKeys(), 2);
|
|
10461
|
-
}
|
|
10462
|
-
function replacerSortKeys() {
|
|
10463
|
-
return (_key, value) => {
|
|
10464
|
-
if (value && typeof value === "object" && !Array.isArray(value)) {
|
|
10465
|
-
const sorted = {};
|
|
10466
|
-
for (const k of Object.keys(value).sort()) {
|
|
10467
|
-
sorted[k] = value[k];
|
|
10468
|
-
}
|
|
10469
|
-
return sorted;
|
|
10470
|
-
}
|
|
10471
|
-
return value;
|
|
10472
|
-
};
|
|
10473
|
-
}
|
|
10474
|
-
|
|
10475
10048
|
// src/profile/index.ts
|
|
10476
10049
|
var profile_exports = {};
|
|
10477
10050
|
__export(profile_exports, {
|
|
@@ -10613,6 +10186,139 @@ function sectionHash(section) {
|
|
|
10613
10186
|
return surfaceContentHash(JSON.stringify({ title: section.title, body: section.body }));
|
|
10614
10187
|
}
|
|
10615
10188
|
|
|
10189
|
+
// src/traced-analyst.ts
|
|
10190
|
+
async function tracedAnalyzeTraces(input, options, traceOpts) {
|
|
10191
|
+
const parentSpan = await traceOpts.emitter.span({
|
|
10192
|
+
kind: "custom",
|
|
10193
|
+
name: "analyst:analyze-traces",
|
|
10194
|
+
parentSpanId: traceOpts.parentSpanId,
|
|
10195
|
+
attributes: {
|
|
10196
|
+
"analyst.question_length": input.question.length,
|
|
10197
|
+
"analyst.max_turns": options.maxTurns ?? 12,
|
|
10198
|
+
"analyst.max_subqueries": options.maxSubqueries ?? 4,
|
|
10199
|
+
"eval.phase": "analyst"
|
|
10200
|
+
}
|
|
10201
|
+
});
|
|
10202
|
+
const originalOnTurn = options.onTurn;
|
|
10203
|
+
const wrappedOptions = {
|
|
10204
|
+
...options,
|
|
10205
|
+
onTurn: async (turn) => {
|
|
10206
|
+
const turnSpan = await traceOpts.emitter.span({
|
|
10207
|
+
kind: "custom",
|
|
10208
|
+
name: `analyst:turn-${turn.turn}`,
|
|
10209
|
+
parentSpanId: parentSpan.span.spanId,
|
|
10210
|
+
attributes: {
|
|
10211
|
+
"analyst.stage": turn.stage,
|
|
10212
|
+
"analyst.turn": turn.turn,
|
|
10213
|
+
"analyst.is_error": turn.isError,
|
|
10214
|
+
"analyst.code_length": turn.code.length,
|
|
10215
|
+
"analyst.output_length": turn.output.length,
|
|
10216
|
+
"eval.phase": "analyst"
|
|
10217
|
+
}
|
|
10218
|
+
});
|
|
10219
|
+
if (turn.isError) {
|
|
10220
|
+
await turnSpan.fail("Turn produced an error");
|
|
10221
|
+
} else {
|
|
10222
|
+
await turnSpan.end();
|
|
10223
|
+
}
|
|
10224
|
+
if (originalOnTurn) await originalOnTurn(turn);
|
|
10225
|
+
}
|
|
10226
|
+
};
|
|
10227
|
+
try {
|
|
10228
|
+
const result = await analyzeTraces(input, wrappedOptions);
|
|
10229
|
+
await parentSpan.end({
|
|
10230
|
+
attributes: {
|
|
10231
|
+
"analyst.question_length": input.question.length,
|
|
10232
|
+
"analyst.turn_count": result.turnCount,
|
|
10233
|
+
"analyst.finding_count": result.findings.length,
|
|
10234
|
+
"analyst.answer_length": result.answer.length,
|
|
10235
|
+
"eval.phase": "analyst"
|
|
10236
|
+
}
|
|
10237
|
+
});
|
|
10238
|
+
return result;
|
|
10239
|
+
} catch (err) {
|
|
10240
|
+
await parentSpan.fail(err instanceof Error ? err : String(err));
|
|
10241
|
+
throw err;
|
|
10242
|
+
}
|
|
10243
|
+
}
|
|
10244
|
+
|
|
10245
|
+
// src/traced-judges.ts
|
|
10246
|
+
function traceJudge(judge, judgeName, opts) {
|
|
10247
|
+
return async (tc, input) => {
|
|
10248
|
+
const span = await opts.emitter.span({
|
|
10249
|
+
kind: "llm",
|
|
10250
|
+
name: `judge:${judgeName}`,
|
|
10251
|
+
parentSpanId: opts.parentSpanId,
|
|
10252
|
+
attributes: {
|
|
10253
|
+
"judge.name": judgeName,
|
|
10254
|
+
"eval.phase": "judge"
|
|
10255
|
+
}
|
|
10256
|
+
});
|
|
10257
|
+
try {
|
|
10258
|
+
const scores2 = await judge(tc, input);
|
|
10259
|
+
const composite = scores2.length > 0 ? scores2.reduce((sum4, s) => sum4 + s.score, 0) / scores2.length : 0;
|
|
10260
|
+
await span.end({
|
|
10261
|
+
attributes: {
|
|
10262
|
+
"judge.name": judgeName,
|
|
10263
|
+
"judge.composite_score": composite,
|
|
10264
|
+
"judge.dimension_count": scores2.length,
|
|
10265
|
+
"eval.phase": "judge"
|
|
10266
|
+
}
|
|
10267
|
+
});
|
|
10268
|
+
return scores2;
|
|
10269
|
+
} catch (err) {
|
|
10270
|
+
await span.fail(err instanceof Error ? err : String(err));
|
|
10271
|
+
throw err;
|
|
10272
|
+
}
|
|
10273
|
+
};
|
|
10274
|
+
}
|
|
10275
|
+
function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
10276
|
+
return async (tc, input) => {
|
|
10277
|
+
const ensembleSpan = await opts.emitter.span({
|
|
10278
|
+
kind: "custom",
|
|
10279
|
+
name: "judge:ensemble",
|
|
10280
|
+
parentSpanId: opts.parentSpanId,
|
|
10281
|
+
attributes: {
|
|
10282
|
+
"judge.ensemble_size": judges.length,
|
|
10283
|
+
"eval.phase": "judge"
|
|
10284
|
+
}
|
|
10285
|
+
});
|
|
10286
|
+
try {
|
|
10287
|
+
const allScores = [];
|
|
10288
|
+
let failedJudges = 0;
|
|
10289
|
+
for (let i = 0; i < judges.length; i++) {
|
|
10290
|
+
const judge = judges[i];
|
|
10291
|
+
const name = judgeNames[i] ?? `judge_${i}`;
|
|
10292
|
+
const tracedFn = traceJudge(judge, name, {
|
|
10293
|
+
emitter: opts.emitter,
|
|
10294
|
+
parentSpanId: ensembleSpan.span.spanId
|
|
10295
|
+
});
|
|
10296
|
+
try {
|
|
10297
|
+
const scores2 = await tracedFn(tc, input);
|
|
10298
|
+
allScores.push(...scores2);
|
|
10299
|
+
} catch (err) {
|
|
10300
|
+
if (!(err instanceof JudgeParseError)) throw err;
|
|
10301
|
+
failedJudges++;
|
|
10302
|
+
}
|
|
10303
|
+
}
|
|
10304
|
+
const composite = allScores.length > 0 ? allScores.reduce((sum4, s) => sum4 + s.score, 0) / allScores.length : 0;
|
|
10305
|
+
await ensembleSpan.end({
|
|
10306
|
+
attributes: {
|
|
10307
|
+
"judge.ensemble_size": judges.length,
|
|
10308
|
+
"judge.composite_score": composite,
|
|
10309
|
+
"judge.total_dimensions": allScores.length,
|
|
10310
|
+
"judge.failed_judges": failedJudges,
|
|
10311
|
+
"eval.phase": "judge"
|
|
10312
|
+
}
|
|
10313
|
+
});
|
|
10314
|
+
return allScores;
|
|
10315
|
+
} catch (err) {
|
|
10316
|
+
await ensembleSpan.fail(err instanceof Error ? err : String(err));
|
|
10317
|
+
throw err;
|
|
10318
|
+
}
|
|
10319
|
+
};
|
|
10320
|
+
}
|
|
10321
|
+
|
|
10616
10322
|
// src/cost-report.ts
|
|
10617
10323
|
function costReport(ledger) {
|
|
10618
10324
|
const summary = ledger.summary();
|
|
@@ -10733,13 +10439,13 @@ function verifyAttestation(report, attested) {
|
|
|
10733
10439
|
}
|
|
10734
10440
|
|
|
10735
10441
|
// src/product-benchmark/index.ts
|
|
10736
|
-
import { existsSync as existsSync6, readFileSync as
|
|
10442
|
+
import { existsSync as existsSync6, readFileSync as readFileSync6, statSync as statSync3 } from "fs";
|
|
10737
10443
|
import { dirname as dirname4, join as join5 } from "path";
|
|
10738
10444
|
|
|
10739
10445
|
// src/product-benchmark/export.ts
|
|
10740
10446
|
import { spawnSync as spawnSync2 } from "child_process";
|
|
10741
10447
|
import { createHash } from "crypto";
|
|
10742
|
-
import { cpSync, existsSync as existsSync5, mkdirSync as mkdirSync3, readFileSync as
|
|
10448
|
+
import { cpSync, existsSync as existsSync5, mkdirSync as mkdirSync3, readFileSync as readFileSync5, writeFileSync } from "fs";
|
|
10743
10449
|
import { basename as basename2, dirname as dirname3, isAbsolute, join as join4, relative, resolve } from "path";
|
|
10744
10450
|
var productBenchmarkMutableSurfaces = [
|
|
10745
10451
|
"prompt",
|
|
@@ -10787,19 +10493,19 @@ function productBenchmarkRepoIdentity() {
|
|
|
10787
10493
|
};
|
|
10788
10494
|
}
|
|
10789
10495
|
function packageVersion(name) {
|
|
10790
|
-
const pkg = JSON.parse(
|
|
10496
|
+
const pkg = JSON.parse(readFileSync5(resolve("package.json"), "utf8"));
|
|
10791
10497
|
if (pkg.name === name && pkg.version) return pkg.version;
|
|
10792
10498
|
const declared = pkg.dependencies?.[name] ?? pkg.devDependencies?.[name];
|
|
10793
10499
|
if (declared) return declared;
|
|
10794
10500
|
const installed = resolve("node_modules", name, "package.json");
|
|
10795
10501
|
if (existsSync5(installed)) {
|
|
10796
|
-
const installedPkg = JSON.parse(
|
|
10502
|
+
const installedPkg = JSON.parse(readFileSync5(installed, "utf8"));
|
|
10797
10503
|
if (installedPkg.version) return installedPkg.version;
|
|
10798
10504
|
}
|
|
10799
10505
|
return "unknown";
|
|
10800
10506
|
}
|
|
10801
10507
|
function readRunRecords(path) {
|
|
10802
|
-
const lines =
|
|
10508
|
+
const lines = readFileSync5(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
|
|
10803
10509
|
return lines.map((line, index) => {
|
|
10804
10510
|
let parsed;
|
|
10805
10511
|
try {
|
|
@@ -11450,7 +11156,7 @@ function productBenchmarkIntegrityFailures(record) {
|
|
|
11450
11156
|
return failures;
|
|
11451
11157
|
}
|
|
11452
11158
|
function readProductBenchmarkRecords(path) {
|
|
11453
|
-
const lines =
|
|
11159
|
+
const lines = readFileSync6(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
|
|
11454
11160
|
const records = [];
|
|
11455
11161
|
for (const [index, line] of lines.entries()) {
|
|
11456
11162
|
try {
|
|
@@ -11463,7 +11169,7 @@ function readProductBenchmarkRecords(path) {
|
|
|
11463
11169
|
}
|
|
11464
11170
|
function readProductBenchmarkManifest(path) {
|
|
11465
11171
|
try {
|
|
11466
|
-
return validateProductBenchmarkManifest(JSON.parse(
|
|
11172
|
+
return validateProductBenchmarkManifest(JSON.parse(readFileSync6(path, "utf8")));
|
|
11467
11173
|
} catch (err) {
|
|
11468
11174
|
wrapValidationError(path, err);
|
|
11469
11175
|
}
|
|
@@ -11666,11 +11372,7 @@ export {
|
|
|
11666
11372
|
OTEL_AGENT_EVAL_SCOPE,
|
|
11667
11373
|
OUTPUT_VALUE,
|
|
11668
11374
|
OtlpFileTraceStore,
|
|
11669
|
-
POLICY_EDIT_AXES,
|
|
11670
|
-
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
11671
|
-
POLICY_EDIT_TARGET_SURFACES,
|
|
11672
11375
|
PairwiseSteeringOptimizer,
|
|
11673
|
-
PolicyEditValidationError,
|
|
11674
11376
|
ProductClient,
|
|
11675
11377
|
PromptRegistry,
|
|
11676
11378
|
REDACTION_VERSION,
|
|
@@ -11716,7 +11418,6 @@ export {
|
|
|
11716
11418
|
ValidationError,
|
|
11717
11419
|
VerificationError,
|
|
11718
11420
|
acquisitionPlansForKnowledgeGaps,
|
|
11719
|
-
admitPolicyEdit,
|
|
11720
11421
|
adversarialJudge,
|
|
11721
11422
|
agentProfileCellHashMaterial,
|
|
11722
11423
|
agentProfileCellKey,
|
|
@@ -11736,7 +11437,6 @@ export {
|
|
|
11736
11437
|
analyzeTraces,
|
|
11737
11438
|
appendScorecard,
|
|
11738
11439
|
applyLlmSpanOtlpAttributes,
|
|
11739
|
-
applyPolicyEditToSurface,
|
|
11740
11440
|
applyToolSpanOtlpAttributes,
|
|
11741
11441
|
argHash,
|
|
11742
11442
|
asNumber,
|
|
@@ -11770,7 +11470,6 @@ export {
|
|
|
11770
11470
|
bootstrapCi,
|
|
11771
11471
|
buildAgentInterfaceProfileCell,
|
|
11772
11472
|
buildAgentProfileCell,
|
|
11773
|
-
buildAgreementJudge,
|
|
11774
11473
|
buildDefaultAnalystRegistry,
|
|
11775
11474
|
buildDriverSystemPrompt,
|
|
11776
11475
|
buildProductBenchmarkManifest,
|
|
@@ -11800,6 +11499,7 @@ export {
|
|
|
11800
11499
|
clamp01,
|
|
11801
11500
|
classifyFailure,
|
|
11802
11501
|
classifyTreatment,
|
|
11502
|
+
claudeCodeSupervisorRunReader,
|
|
11803
11503
|
cliffsDelta,
|
|
11804
11504
|
clusteredPairedBinary,
|
|
11805
11505
|
codeExecutionJudge,
|
|
@@ -11817,7 +11517,6 @@ export {
|
|
|
11817
11517
|
composeValidators,
|
|
11818
11518
|
computeExperimentStats,
|
|
11819
11519
|
computeFindingId,
|
|
11820
|
-
computePolicyEditId,
|
|
11821
11520
|
computeToolUseMetrics,
|
|
11822
11521
|
computeTraceMetrics,
|
|
11823
11522
|
confidenceInterval,
|
|
@@ -11864,10 +11563,8 @@ export {
|
|
|
11864
11563
|
defaultBlendWeights,
|
|
11865
11564
|
defaultIsMaterial,
|
|
11866
11565
|
defaultJudges,
|
|
11867
|
-
defaultParseStudentLabel,
|
|
11868
11566
|
defaultProviderRedactor,
|
|
11869
11567
|
defaultReferenceReplayMatcher,
|
|
11870
|
-
defaultRenderStudentPrompt,
|
|
11871
11568
|
defaultTraceInsightPanel,
|
|
11872
11569
|
deployGateLayer,
|
|
11873
11570
|
describeTraceInsightScope,
|
|
@@ -11907,7 +11604,6 @@ export {
|
|
|
11907
11604
|
feedbackTrajectoriesToOptimizerRows,
|
|
11908
11605
|
feedbackTrajectoryToDatasetScenario,
|
|
11909
11606
|
feedbackTrajectoryToOptimizerRow,
|
|
11910
|
-
fieldAgreement,
|
|
11911
11607
|
fileContains,
|
|
11912
11608
|
fileExists,
|
|
11913
11609
|
fileExperimentStore,
|
|
@@ -11965,7 +11661,6 @@ export {
|
|
|
11965
11661
|
isLlmSpan,
|
|
11966
11662
|
isModelPriced,
|
|
11967
11663
|
isOtelConfigured,
|
|
11968
|
-
isPolicyEdit,
|
|
11969
11664
|
isRetrievalSpan,
|
|
11970
11665
|
isRolloutLine,
|
|
11971
11666
|
isRunRecord,
|
|
@@ -11991,15 +11686,12 @@ export {
|
|
|
11991
11686
|
llmJudge,
|
|
11992
11687
|
llmSpanFromProvider,
|
|
11993
11688
|
llmSpans,
|
|
11994
|
-
loadGoldScenarios,
|
|
11995
11689
|
loadScorecard,
|
|
11996
11690
|
loadScorerFromGrader,
|
|
11997
11691
|
localCommandRunner,
|
|
11998
11692
|
lowercaseMutator,
|
|
11999
11693
|
makeEvalTools,
|
|
12000
11694
|
makeFinding,
|
|
12001
|
-
makePolicyEdit,
|
|
12002
|
-
makePolicyEditCandidateRecord,
|
|
12003
11695
|
mannWhitneyU,
|
|
12004
11696
|
mapConcurrent,
|
|
12005
11697
|
matchGoldens,
|
|
@@ -12040,7 +11732,6 @@ export {
|
|
|
12040
11732
|
paretoFrontierWithCrowding,
|
|
12041
11733
|
parseCorrectnessResponse,
|
|
12042
11734
|
parseFeedbackTrajectoriesJsonl,
|
|
12043
|
-
parseGoldJsonl,
|
|
12044
11735
|
parseReflectionResponse,
|
|
12045
11736
|
parseRunRecordSafe,
|
|
12046
11737
|
parseRuntimeTrajectoryHookEvent,
|
|
@@ -12051,8 +11742,6 @@ export {
|
|
|
12051
11742
|
pearsonR,
|
|
12052
11743
|
pixelDeltaRatio,
|
|
12053
11744
|
planTraceInsightQuestions,
|
|
12054
|
-
policyEditFromFinding,
|
|
12055
|
-
policyEditsFromFindings,
|
|
12056
11745
|
politenessPrefixMutator,
|
|
12057
11746
|
positionalBias,
|
|
12058
11747
|
preflightModels,
|
|
@@ -12070,6 +11759,7 @@ export {
|
|
|
12070
11759
|
providerFromBaseUrl,
|
|
12071
11760
|
pytestTestParser,
|
|
12072
11761
|
ranks,
|
|
11762
|
+
readClaudeCodeSupervisorRun,
|
|
12073
11763
|
readOtlpStatus,
|
|
12074
11764
|
readProductBenchmarkManifest,
|
|
12075
11765
|
readProductBenchmarkRecords,
|
|
@@ -12114,7 +11804,6 @@ export {
|
|
|
12114
11804
|
runBehavioralCanaries,
|
|
12115
11805
|
runCanaries,
|
|
12116
11806
|
runCounterfactual,
|
|
12117
|
-
runDistillation,
|
|
12118
11807
|
runE2EWorkflow,
|
|
12119
11808
|
runEvalCampaign,
|
|
12120
11809
|
runExpectations,
|
|
@@ -12140,7 +11829,6 @@ export {
|
|
|
12140
11829
|
scoreContinuity,
|
|
12141
11830
|
scoreFromEvals,
|
|
12142
11831
|
scoreKnowledgeReadiness,
|
|
12143
|
-
scorePolicyEditReadiness,
|
|
12144
11832
|
scorePrReviewComments,
|
|
12145
11833
|
scorePrReviewSource,
|
|
12146
11834
|
scoreRedTeamOutput,
|
|
@@ -12155,7 +11843,6 @@ export {
|
|
|
12155
11843
|
showMeasured,
|
|
12156
11844
|
signManifest,
|
|
12157
11845
|
spearmanR,
|
|
12158
|
-
splitGold,
|
|
12159
11846
|
statusAdvanced,
|
|
12160
11847
|
stopOnNoProgress,
|
|
12161
11848
|
stopOnRepeatedAction,
|
|
@@ -12193,8 +11880,6 @@ export {
|
|
|
12193
11880
|
urlContains,
|
|
12194
11881
|
userQuestionsForKnowledgeGaps,
|
|
12195
11882
|
validateAgentProfileCell,
|
|
12196
|
-
validatePolicyEdit,
|
|
12197
|
-
validatePolicyEditCandidateRecord,
|
|
12198
11883
|
validateProductBenchmarkManifest,
|
|
12199
11884
|
validateProductBenchmarkRecord,
|
|
12200
11885
|
validateProductBenchmarkRun,
|