@tangle-network/agent-eval 0.86.0 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +3 -3
- package/dist/adapters/langchain.d.ts +3 -3
- package/dist/adapters/otel.d.ts +6 -6
- package/dist/adversarial-DIVcDoI_.d.ts +88 -0
- package/dist/analyst/index.d.ts +11 -10
- package/dist/analyst/index.js +13 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
- package/dist/belief-state/index.d.ts +4 -4
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +3 -3
- package/dist/campaign/index.d.ts +165 -18
- package/dist/campaign/index.js +289 -14
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-45EEMHTC.js +35 -0
- package/dist/chunk-45EEMHTC.js.map +1 -0
- package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
- package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
- package/dist/chunk-5HRORJQY.js.map +1 -0
- package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
- package/dist/chunk-BOD4O7OF.js +40 -0
- package/dist/chunk-BOD4O7OF.js.map +1 -0
- package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
- package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
- package/dist/chunk-D3V5B42D.js.map +1 -0
- package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
- package/dist/chunk-FIUKOSWI.js.map +1 -0
- package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
- package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
- package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
- package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
- package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
- package/dist/chunk-QAY5UIJO.js.map +1 -0
- package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
- package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
- package/dist/chunk-REVYNR6C.js +100 -0
- package/dist/chunk-REVYNR6C.js.map +1 -0
- package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
- package/dist/chunk-UHMJT4T7.js +200 -0
- package/dist/chunk-UHMJT4T7.js.map +1 -0
- package/dist/chunk-UMMZHCPB.js +190 -0
- package/dist/chunk-UMMZHCPB.js.map +1 -0
- package/dist/chunk-VZSRQ272.js +149 -0
- package/dist/chunk-VZSRQ272.js.map +1 -0
- package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
- package/dist/chunk-XY4DDNEG.js.map +1 -0
- package/dist/chunk-Y47J2LJ3.js +859 -0
- package/dist/chunk-Y47J2LJ3.js.map +1 -0
- package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
- package/dist/chunk-ZFIBGEOL.js.map +1 -0
- package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
- package/dist/contract/index.d.ts +24 -95
- package/dist/contract/index.js +16 -755
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
- package/dist/control.d.ts +5 -5
- package/dist/corpus-BoR-041R.d.ts +560 -0
- package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
- package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
- package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
- package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
- package/dist/diagnose.d.ts +251 -0
- package/dist/diagnose.js +381 -0
- package/dist/diagnose.js.map +1 -0
- package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
- package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
- package/dist/fuzz.d.ts +484 -0
- package/dist/fuzz.js +613 -0
- package/dist/fuzz.js.map +1 -0
- package/dist/governance/index.d.ts +4 -4
- package/dist/hosted/index.d.ts +6 -6
- package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
- package/dist/index.d.ts +717 -455
- package/dist/index.js +1590 -793
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
- package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
- package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
- package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
- package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +177 -3
- package/dist/meta-eval/index.js +260 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
- package/dist/multishot/index.d.ts +25 -11
- package/dist/multishot/index.js +36 -7
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
- package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
- package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
- package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/reporting.js +3 -3
- package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
- package/dist/rl.d.ts +194 -656
- package/dist/rl.js +236 -154
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
- package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
- package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
- package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
- package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
- package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
- package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
- package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
- package/dist/traces.d.ts +3 -3
- package/dist/traces.js +8 -6
- package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
- package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
- package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
- package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/workflow/index.d.ts +12 -11
- package/dist/workflow/index.js +1 -1
- package/package.json +11 -1
- package/dist/chunk-BABOZOSN.js.map +0 -1
- package/dist/chunk-L5G7OUKD.js.map +0 -1
- package/dist/chunk-SHTXZ4O2.js +0 -113
- package/dist/chunk-SHTXZ4O2.js.map +0 -1
- package/dist/chunk-VIDQF3F5.js.map +0 -1
- package/dist/chunk-VUINJM5M.js.map +0 -1
- package/dist/chunk-YGYXHNAQ.js.map +0 -1
- package/dist/chunk-YV7J7X5N.js.map +0 -1
- /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
- /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
- /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
- /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
- /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
- /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
- /package/dist/{run-campaign-4Y5V5CN3.js.map → run-campaign-RDGAM5KJ.js.map} +0 -0
package/dist/index.js
CHANGED
|
@@ -1,22 +1,35 @@
|
|
|
1
|
+
import {
|
|
2
|
+
HoldoutAuditor,
|
|
3
|
+
analyzeRuns,
|
|
4
|
+
canaryLeakView,
|
|
5
|
+
checkBehavioralCanary,
|
|
6
|
+
checkCanaries,
|
|
7
|
+
runBehavioralCanaries
|
|
8
|
+
} from "./chunk-Y47J2LJ3.js";
|
|
9
|
+
import {
|
|
10
|
+
classifyEuAiRisk,
|
|
11
|
+
euAiActReport,
|
|
12
|
+
nistAiRmfReport,
|
|
13
|
+
renderMarkdown,
|
|
14
|
+
soc2Report,
|
|
15
|
+
summarize
|
|
16
|
+
} from "./chunk-KKHDIONI.js";
|
|
17
|
+
import {
|
|
18
|
+
acquisitionPlansForKnowledgeGaps,
|
|
19
|
+
blockingKnowledgeEval,
|
|
20
|
+
knowledgeReadinessTracePayload,
|
|
21
|
+
scoreKnowledgeReadiness,
|
|
22
|
+
userQuestionsForKnowledgeGaps
|
|
23
|
+
} from "./chunk-3CKU6VGU.js";
|
|
1
24
|
import {
|
|
2
25
|
agentProfileHash,
|
|
26
|
+
completionVerdict,
|
|
3
27
|
createLlmCorrectnessChecker,
|
|
4
28
|
createTokenRecallChecker,
|
|
5
29
|
extractProducedState,
|
|
6
30
|
parseCorrectnessResponse,
|
|
7
31
|
verifyCompletion
|
|
8
|
-
} from "./chunk-
|
|
9
|
-
import {
|
|
10
|
-
parseRuntimeTrajectoryHookEvent,
|
|
11
|
-
projectRuntimeTrajectoryEvidence
|
|
12
|
-
} from "./chunk-T4SQEITX.js";
|
|
13
|
-
import {
|
|
14
|
-
HoldoutAuditor,
|
|
15
|
-
canaryLeakView,
|
|
16
|
-
checkBehavioralCanary,
|
|
17
|
-
checkCanaries,
|
|
18
|
-
runBehavioralCanaries
|
|
19
|
-
} from "./chunk-SHTXZ4O2.js";
|
|
32
|
+
} from "./chunk-FIUKOSWI.js";
|
|
20
33
|
import {
|
|
21
34
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
22
35
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -40,12 +53,16 @@ import {
|
|
|
40
53
|
scoreRedTeamOutput,
|
|
41
54
|
surfaceContentHash,
|
|
42
55
|
toolNamesForRun
|
|
43
|
-
} from "./chunk-
|
|
56
|
+
} from "./chunk-ZFIBGEOL.js";
|
|
44
57
|
import {
|
|
45
58
|
BackendIntegrityError,
|
|
46
59
|
assertRealBackend,
|
|
47
60
|
summarizeBackendIntegrity
|
|
48
|
-
} from "./chunk-
|
|
61
|
+
} from "./chunk-TWS7AZEY.js";
|
|
62
|
+
import {
|
|
63
|
+
parseRuntimeTrajectoryHookEvent,
|
|
64
|
+
projectRuntimeTrajectoryEvidence
|
|
65
|
+
} from "./chunk-T4SQEITX.js";
|
|
49
66
|
import {
|
|
50
67
|
MODEL_PRICING,
|
|
51
68
|
MetricsCollector,
|
|
@@ -67,14 +84,14 @@ import {
|
|
|
67
84
|
computeToolUseMetrics,
|
|
68
85
|
iqr,
|
|
69
86
|
welchsTTest
|
|
70
|
-
} from "./chunk-
|
|
87
|
+
} from "./chunk-L3JOU6XM.js";
|
|
88
|
+
import {
|
|
89
|
+
analyzeSeries
|
|
90
|
+
} from "./chunk-BOD4O7OF.js";
|
|
71
91
|
import {
|
|
72
92
|
exportTrainingData,
|
|
73
93
|
toNdjson
|
|
74
94
|
} from "./chunk-KMPRBJK4.js";
|
|
75
|
-
import {
|
|
76
|
-
buildTrajectory
|
|
77
|
-
} from "./chunk-RZTMDUO7.js";
|
|
78
95
|
import {
|
|
79
96
|
DockerSandboxDriver,
|
|
80
97
|
SandboxHarness,
|
|
@@ -85,21 +102,6 @@ import {
|
|
|
85
102
|
runTestGradedScenario,
|
|
86
103
|
vitestTestParser
|
|
87
104
|
} from "./chunk-T375SUOZ.js";
|
|
88
|
-
import {
|
|
89
|
-
classifyEuAiRisk,
|
|
90
|
-
euAiActReport,
|
|
91
|
-
nistAiRmfReport,
|
|
92
|
-
renderMarkdown,
|
|
93
|
-
soc2Report,
|
|
94
|
-
summarize
|
|
95
|
-
} from "./chunk-KKHDIONI.js";
|
|
96
|
-
import {
|
|
97
|
-
acquisitionPlansForKnowledgeGaps,
|
|
98
|
-
blockingKnowledgeEval,
|
|
99
|
-
knowledgeReadinessTracePayload,
|
|
100
|
-
scoreKnowledgeReadiness,
|
|
101
|
-
userQuestionsForKnowledgeGaps
|
|
102
|
-
} from "./chunk-3CKU6VGU.js";
|
|
103
105
|
import {
|
|
104
106
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
105
107
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
@@ -111,9 +113,7 @@ import {
|
|
|
111
113
|
SKILL_USAGE_ANALYST,
|
|
112
114
|
SkillUsageAnalyst,
|
|
113
115
|
aggregateRunScore,
|
|
114
|
-
buildDefaultAnalystRegistry,
|
|
115
116
|
clamp01,
|
|
116
|
-
computeTraceMetrics,
|
|
117
117
|
createAnalystAi,
|
|
118
118
|
createChatClient,
|
|
119
119
|
createSemanticConceptJudge,
|
|
@@ -121,7 +121,11 @@ import {
|
|
|
121
121
|
diffFindings,
|
|
122
122
|
resetLockedAppendersForTesting,
|
|
123
123
|
runSemanticConceptJudge
|
|
124
|
-
} from "./chunk-
|
|
124
|
+
} from "./chunk-XY4DDNEG.js";
|
|
125
|
+
import {
|
|
126
|
+
buildDefaultAnalystRegistry,
|
|
127
|
+
computeTraceMetrics
|
|
128
|
+
} from "./chunk-UMMZHCPB.js";
|
|
125
129
|
import {
|
|
126
130
|
AnalystRegistry,
|
|
127
131
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -129,11 +133,9 @@ import {
|
|
|
129
133
|
IMPROVEMENT_KIND_SPEC,
|
|
130
134
|
KNOWLEDGE_GAP_KIND_SPEC,
|
|
131
135
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
132
|
-
computeFindingId,
|
|
133
136
|
createTraceAnalystKind,
|
|
134
|
-
makeFinding,
|
|
135
137
|
renderPriorFindings
|
|
136
|
-
} from "./chunk-
|
|
138
|
+
} from "./chunk-D3V5B42D.js";
|
|
137
139
|
import {
|
|
138
140
|
controlFailureClassFromVerification,
|
|
139
141
|
controlRunToRunRecord,
|
|
@@ -159,10 +161,10 @@ import {
|
|
|
159
161
|
evaluateReleaseConfidence,
|
|
160
162
|
judgeReplayGate,
|
|
161
163
|
renderReleaseReport
|
|
162
|
-
} from "./chunk-
|
|
164
|
+
} from "./chunk-4FBZZIYD.js";
|
|
163
165
|
import {
|
|
164
166
|
runEvalCampaign
|
|
165
|
-
} from "./chunk-
|
|
167
|
+
} from "./chunk-GSH6QNNS.js";
|
|
166
168
|
import {
|
|
167
169
|
LlmCallError,
|
|
168
170
|
LlmClient,
|
|
@@ -185,7 +187,18 @@ import {
|
|
|
185
187
|
paretoChart,
|
|
186
188
|
researchReport,
|
|
187
189
|
summaryTable
|
|
188
|
-
} from "./chunk-
|
|
190
|
+
} from "./chunk-CY6U5S3X.js";
|
|
191
|
+
import {
|
|
192
|
+
attributeCounterfactuals,
|
|
193
|
+
runCounterfactual
|
|
194
|
+
} from "./chunk-REVYNR6C.js";
|
|
195
|
+
import {
|
|
196
|
+
buildTrajectory
|
|
197
|
+
} from "./chunk-RZTMDUO7.js";
|
|
198
|
+
import {
|
|
199
|
+
computeFindingId,
|
|
200
|
+
makeFinding
|
|
201
|
+
} from "./chunk-45EEMHTC.js";
|
|
189
202
|
import {
|
|
190
203
|
benjaminiHochberg,
|
|
191
204
|
bonferroni,
|
|
@@ -197,9 +210,11 @@ import {
|
|
|
197
210
|
continuousAgreement,
|
|
198
211
|
corpusInterRaterAgreement,
|
|
199
212
|
corpusInterRaterAgreementFromJudgeScores,
|
|
213
|
+
eProcess,
|
|
200
214
|
interRaterReliability,
|
|
201
215
|
interpretCliffs,
|
|
202
216
|
mannWhitneyU,
|
|
217
|
+
mulberry32,
|
|
203
218
|
normalizeScores,
|
|
204
219
|
pairedBootstrap,
|
|
205
220
|
pairedMde,
|
|
@@ -212,7 +227,7 @@ import {
|
|
|
212
227
|
weightedComposite,
|
|
213
228
|
weightedMean,
|
|
214
229
|
wilcoxonSignedRank
|
|
215
|
-
} from "./chunk-
|
|
230
|
+
} from "./chunk-LMZQ2Z4U.js";
|
|
216
231
|
import {
|
|
217
232
|
FileSystemTraceStore,
|
|
218
233
|
InMemoryTraceStore,
|
|
@@ -243,7 +258,7 @@ import {
|
|
|
243
258
|
scoreTraceInsightReadiness,
|
|
244
259
|
tokenizeDomainWords,
|
|
245
260
|
traceAnalystOnRunComplete
|
|
246
|
-
} from "./chunk-
|
|
261
|
+
} from "./chunk-QG2OVF2D.js";
|
|
247
262
|
import {
|
|
248
263
|
DEFAULT_REDACTION_RULES,
|
|
249
264
|
REDACTION_VERSION,
|
|
@@ -270,16 +285,18 @@ import {
|
|
|
270
285
|
isToolSpan
|
|
271
286
|
} from "./chunk-5BKGXME7.js";
|
|
272
287
|
import {
|
|
273
|
-
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
274
|
-
OtlpFileTraceStore,
|
|
275
|
-
SpanNotFoundError,
|
|
276
288
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
277
289
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
278
290
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
291
|
+
analyzeTraces
|
|
292
|
+
} from "./chunk-UHMJT4T7.js";
|
|
293
|
+
import {
|
|
294
|
+
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
295
|
+
OtlpFileTraceStore,
|
|
296
|
+
SpanNotFoundError,
|
|
279
297
|
TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
|
|
280
298
|
TraceFileMissingError,
|
|
281
299
|
TraceNotFoundError,
|
|
282
|
-
analyzeTraces,
|
|
283
300
|
asNumber,
|
|
284
301
|
asString,
|
|
285
302
|
buildTraceAnalystTools,
|
|
@@ -291,7 +308,7 @@ import {
|
|
|
291
308
|
readOtlpStatus,
|
|
292
309
|
stringField,
|
|
293
310
|
traceAnalystFunctionGroup
|
|
294
|
-
} from "./chunk-
|
|
311
|
+
} from "./chunk-QAY5UIJO.js";
|
|
295
312
|
import {
|
|
296
313
|
RunIntegrityError,
|
|
297
314
|
assertRunCaptured,
|
|
@@ -646,132 +663,337 @@ function renderCommitMessage(input) {
|
|
|
646
663
|
return lines.join("\n").trim();
|
|
647
664
|
}
|
|
648
665
|
|
|
649
|
-
// src/
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
666
|
+
// src/judges.ts
|
|
667
|
+
var JudgeParseError = class extends JudgeError {
|
|
668
|
+
/** Name of the judge whose response failed to parse. */
|
|
669
|
+
judgeName;
|
|
670
|
+
/** The raw (truncated) model response that failed to parse. */
|
|
671
|
+
raw;
|
|
672
|
+
constructor(judgeName, raw, options) {
|
|
673
|
+
super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
|
|
674
|
+
this.judgeName = judgeName;
|
|
675
|
+
this.raw = raw;
|
|
676
|
+
}
|
|
677
|
+
};
|
|
678
|
+
function createDomainExpertJudge(domain) {
|
|
679
|
+
return async (tc, { scenario, turns }) => {
|
|
680
|
+
const conversation = turns.map(
|
|
681
|
+
(t, i) => `Turn ${i + 1}:
|
|
682
|
+
User: ${t.userMessage}
|
|
683
|
+
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
684
|
+
).join("\n\n---\n\n");
|
|
664
685
|
const resp = await tc.chat({
|
|
665
|
-
model,
|
|
666
|
-
messages
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
if (idx > 0) fields[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
|
|
686
|
-
}
|
|
687
|
-
const blockType = blockMatch[1] ?? "";
|
|
688
|
-
allBlocks.push({ type: blockType, fields });
|
|
689
|
-
turnBlocks.push({ type: blockType, title: fields.title ?? "" });
|
|
690
|
-
blockMatch = blockReLocal.exec(content);
|
|
691
|
-
}
|
|
692
|
-
let hasToolCall = false;
|
|
693
|
-
if (config.toolCallPatterns) {
|
|
694
|
-
for (const pattern of config.toolCallPatterns) {
|
|
695
|
-
const re = new RegExp(pattern.source, pattern.flags);
|
|
696
|
-
let toolMatch = re.exec(content);
|
|
697
|
-
while (toolMatch !== null) {
|
|
698
|
-
allToolCalls.push(toolMatch[0]);
|
|
699
|
-
hasToolCall = true;
|
|
700
|
-
toolMatch = re.exec(content);
|
|
686
|
+
model: "gpt-4o",
|
|
687
|
+
messages: [
|
|
688
|
+
{
|
|
689
|
+
role: "system",
|
|
690
|
+
content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
|
|
691
|
+
|
|
692
|
+
Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
|
|
693
|
+
|
|
694
|
+
Evaluate:
|
|
695
|
+
1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
|
|
696
|
+
2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
|
|
697
|
+
|
|
698
|
+
Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
|
|
699
|
+
},
|
|
700
|
+
{
|
|
701
|
+
role: "user",
|
|
702
|
+
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
703
|
+
Scenario: ${scenario.thesis}
|
|
704
|
+
|
|
705
|
+
${conversation}`
|
|
701
706
|
}
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
turnIndex: i,
|
|
706
|
-
userMessage: turn.user,
|
|
707
|
-
agentResponse: content,
|
|
708
|
-
durationMs: Date.now() - turnStart,
|
|
709
|
-
blocksExtracted: turnBlocks,
|
|
710
|
-
containsCode: allCodeBlocks.length > 0,
|
|
711
|
-
containsToolCall: hasToolCall
|
|
707
|
+
],
|
|
708
|
+
temperature: 0.1,
|
|
709
|
+
maxTokens: 800
|
|
712
710
|
});
|
|
713
|
-
|
|
714
|
-
const artifacts = {
|
|
715
|
-
vaultFiles: [],
|
|
716
|
-
blocksExtracted: allBlocks,
|
|
717
|
-
codeBlocks: allCodeBlocks,
|
|
718
|
-
toolCalls: allToolCalls
|
|
711
|
+
return parseJudgeResponse("domain_expert", resp);
|
|
719
712
|
};
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
passed: count >= (check2.minCount ?? 1),
|
|
731
|
-
detail: `Found ${count} ${check2.target} blocks (need ${check2.minCount ?? 1})`
|
|
732
|
-
};
|
|
713
|
+
}
|
|
714
|
+
var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
|
|
715
|
+
const codeBlocks = artifacts.codeBlocks;
|
|
716
|
+
if (codeBlocks.length === 0) {
|
|
717
|
+
return [
|
|
718
|
+
{
|
|
719
|
+
judgeName: "code_execution",
|
|
720
|
+
dimension: "code_execution",
|
|
721
|
+
score: 0,
|
|
722
|
+
reasoning: "No code blocks found in agent response."
|
|
733
723
|
}
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
724
|
+
];
|
|
725
|
+
}
|
|
726
|
+
const codeText = codeBlocks.map(
|
|
727
|
+
(b, i) => `Block ${i + 1} (${b.language}):
|
|
728
|
+
\`\`\`${b.language}
|
|
729
|
+
${b.code.slice(0, 3e3)}
|
|
730
|
+
\`\`\``
|
|
731
|
+
).join("\n\n");
|
|
732
|
+
const resp = await tc.chat({
|
|
733
|
+
model: "gpt-4o",
|
|
734
|
+
messages: [
|
|
735
|
+
{
|
|
736
|
+
role: "system",
|
|
737
|
+
content: `You are a principal software engineer reviewing code written by an AI agent.
|
|
738
|
+
|
|
739
|
+
Score STRICTLY:
|
|
740
|
+
1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
|
|
741
|
+
2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
|
|
742
|
+
3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
|
|
743
|
+
|
|
744
|
+
Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
|
|
745
|
+
},
|
|
746
|
+
{
|
|
747
|
+
role: "user",
|
|
748
|
+
content: `Task: ${scenario.thesis}
|
|
749
|
+
|
|
750
|
+
${codeText}`
|
|
739
751
|
}
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
passed: false,
|
|
744
|
-
detail: `Check type "${check2.type}" requires live environment`
|
|
745
|
-
};
|
|
746
|
-
}
|
|
752
|
+
],
|
|
753
|
+
temperature: 0.1,
|
|
754
|
+
maxTokens: 1e3
|
|
747
755
|
});
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
756
|
+
return parseJudgeResponse("code_execution", resp);
|
|
757
|
+
};
|
|
758
|
+
var coherenceJudge = async (tc, { scenario, turns }) => {
|
|
759
|
+
if (turns.length < 2) {
|
|
760
|
+
return [];
|
|
761
|
+
}
|
|
762
|
+
const conversation = turns.map(
|
|
763
|
+
(t, i) => `Turn ${i + 1}:
|
|
764
|
+
User: ${t.userMessage}
|
|
765
|
+
Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
|
|
766
|
+
).join("\n\n---\n\n");
|
|
767
|
+
const resp = await tc.chat({
|
|
768
|
+
model: "gpt-4o",
|
|
769
|
+
messages: [
|
|
770
|
+
{
|
|
771
|
+
role: "system",
|
|
772
|
+
content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
|
|
773
|
+
|
|
774
|
+
Score STRICTLY:
|
|
775
|
+
1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
|
|
776
|
+
2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
|
|
777
|
+
3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
|
|
778
|
+
|
|
779
|
+
Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
|
|
780
|
+
},
|
|
781
|
+
{
|
|
782
|
+
role: "user",
|
|
783
|
+
content: `Scenario: ${scenario.thesis}
|
|
784
|
+
|
|
785
|
+
${conversation}`
|
|
786
|
+
}
|
|
787
|
+
],
|
|
788
|
+
temperature: 0.1,
|
|
789
|
+
maxTokens: 800
|
|
790
|
+
});
|
|
791
|
+
return parseJudgeResponse("coherence", resp);
|
|
792
|
+
};
|
|
793
|
+
var adversarialJudge = async (tc, { scenario, turns }) => {
|
|
794
|
+
const conversation = turns.map(
|
|
795
|
+
(t, i) => `Turn ${i + 1}:
|
|
796
|
+
User: ${t.userMessage}
|
|
797
|
+
Agent: ${t.agentResponse.slice(0, 1500)}`
|
|
798
|
+
).join("\n\n---\n\n");
|
|
799
|
+
const resp = await tc.chat({
|
|
800
|
+
model: "gpt-4o",
|
|
801
|
+
messages: [
|
|
802
|
+
{
|
|
803
|
+
role: "system",
|
|
804
|
+
content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
|
|
805
|
+
|
|
806
|
+
1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
|
|
807
|
+
2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
|
|
808
|
+
3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
|
|
809
|
+
|
|
810
|
+
Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
|
|
811
|
+
|
|
812
|
+
Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
|
|
813
|
+
},
|
|
814
|
+
{
|
|
815
|
+
role: "user",
|
|
816
|
+
content: `Persona: ${scenario.persona}
|
|
817
|
+
Scenario: ${scenario.thesis}
|
|
818
|
+
|
|
819
|
+
${conversation}`
|
|
820
|
+
}
|
|
821
|
+
],
|
|
822
|
+
temperature: 0.2,
|
|
823
|
+
maxTokens: 800
|
|
824
|
+
});
|
|
825
|
+
return parseJudgeResponse("adversarial", resp);
|
|
826
|
+
};
|
|
827
|
+
function createCustomJudge(name, systemPrompt, opts) {
|
|
828
|
+
return async (tc, { scenario, turns }) => {
|
|
829
|
+
const conversation = turns.map(
|
|
830
|
+
(t, i) => `Turn ${i + 1}:
|
|
831
|
+
User: ${t.userMessage}
|
|
832
|
+
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
833
|
+
).join("\n\n---\n\n");
|
|
834
|
+
const resp = await tc.chat({
|
|
835
|
+
model: opts?.model ?? "gpt-4o",
|
|
836
|
+
messages: [
|
|
837
|
+
{
|
|
838
|
+
role: "system",
|
|
839
|
+
content: systemPrompt
|
|
840
|
+
},
|
|
841
|
+
{
|
|
842
|
+
role: "user",
|
|
843
|
+
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
844
|
+
Scenario: ${scenario.thesis}
|
|
845
|
+
|
|
846
|
+
${conversation}`
|
|
847
|
+
}
|
|
848
|
+
],
|
|
849
|
+
temperature: opts?.temperature ?? 0.1,
|
|
850
|
+
maxTokens: opts?.maxTokens ?? 1e3
|
|
851
|
+
});
|
|
852
|
+
return parseJudgeResponse(name, resp);
|
|
853
|
+
};
|
|
854
|
+
}
|
|
855
|
+
function defaultJudges(domain) {
|
|
856
|
+
return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
|
|
857
|
+
}
|
|
858
|
+
function parseJudgeResponse(judgeName, resp) {
|
|
859
|
+
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
860
|
+
try {
|
|
861
|
+
let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
|
|
862
|
+
const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
|
|
863
|
+
if (arrayMatch) cleaned = arrayMatch[0];
|
|
864
|
+
const parsed = JSON.parse(cleaned);
|
|
865
|
+
return parsed.map((p) => ({
|
|
866
|
+
judgeName,
|
|
867
|
+
dimension: p.dimension,
|
|
868
|
+
score: Math.max(0, Math.min(10, p.score)),
|
|
869
|
+
reasoning: p.reasoning ?? "",
|
|
870
|
+
evidence: p.evidence
|
|
871
|
+
}));
|
|
872
|
+
} catch (err) {
|
|
873
|
+
throw new JudgeParseError(judgeName, content, { cause: err });
|
|
874
|
+
}
|
|
875
|
+
}
|
|
876
|
+
|
|
877
|
+
// src/executor.ts
|
|
878
|
+
async function executeScenario(tc, scenario, config) {
|
|
879
|
+
const startTime = Date.now();
|
|
880
|
+
const model = config.model ?? "gpt-4o";
|
|
881
|
+
const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
|
|
882
|
+
const messages = [{ role: "system", content: systemPrompt }];
|
|
883
|
+
const turns = [];
|
|
884
|
+
const allCodeBlocks = [];
|
|
885
|
+
const allBlocks = [];
|
|
886
|
+
const allToolCalls = [];
|
|
887
|
+
const blockRe = config.blockPattern ?? /:::(\w+)\s*\n([\s\S]*?)\n\s*:::/g;
|
|
888
|
+
for (let i = 0; i < scenario.turns.length; i++) {
|
|
889
|
+
const turn = scenario.turns[i];
|
|
890
|
+
const turnStart = Date.now();
|
|
891
|
+
messages.push({ role: "user", content: turn.user });
|
|
892
|
+
const resp = await tc.chat({
|
|
893
|
+
model,
|
|
894
|
+
messages,
|
|
895
|
+
temperature: 0.4,
|
|
896
|
+
maxTokens: 3e3
|
|
897
|
+
});
|
|
898
|
+
const content = resp.choices?.[0]?.message?.content ?? "";
|
|
899
|
+
messages.push({ role: "assistant", content });
|
|
900
|
+
const codeRe = /```(\w+)?\n([\s\S]*?)```/g;
|
|
901
|
+
let codeMatch = codeRe.exec(content);
|
|
902
|
+
while (codeMatch !== null) {
|
|
903
|
+
allCodeBlocks.push({ language: codeMatch[1] ?? "text", code: codeMatch[2] ?? "" });
|
|
904
|
+
codeMatch = codeRe.exec(content);
|
|
905
|
+
}
|
|
906
|
+
const turnBlocks = [];
|
|
907
|
+
const blockReLocal = new RegExp(blockRe.source, blockRe.flags);
|
|
908
|
+
let blockMatch = blockReLocal.exec(content);
|
|
909
|
+
while (blockMatch !== null) {
|
|
910
|
+
const fields = {};
|
|
911
|
+
for (const line of (blockMatch[2] ?? "").split("\n")) {
|
|
912
|
+
const idx = line.indexOf(":");
|
|
913
|
+
if (idx > 0) fields[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
|
|
914
|
+
}
|
|
915
|
+
const blockType = blockMatch[1] ?? "";
|
|
916
|
+
allBlocks.push({ type: blockType, fields });
|
|
917
|
+
turnBlocks.push({ type: blockType, title: fields.title ?? "" });
|
|
918
|
+
blockMatch = blockReLocal.exec(content);
|
|
919
|
+
}
|
|
920
|
+
let hasToolCall = false;
|
|
921
|
+
if (config.toolCallPatterns) {
|
|
922
|
+
for (const pattern of config.toolCallPatterns) {
|
|
923
|
+
const re = new RegExp(pattern.source, pattern.flags);
|
|
924
|
+
let toolMatch = re.exec(content);
|
|
925
|
+
while (toolMatch !== null) {
|
|
926
|
+
allToolCalls.push(toolMatch[0]);
|
|
927
|
+
hasToolCall = true;
|
|
928
|
+
toolMatch = re.exec(content);
|
|
929
|
+
}
|
|
930
|
+
}
|
|
931
|
+
}
|
|
932
|
+
turns.push({
|
|
933
|
+
turnIndex: i,
|
|
934
|
+
userMessage: turn.user,
|
|
935
|
+
agentResponse: content,
|
|
936
|
+
durationMs: Date.now() - turnStart,
|
|
937
|
+
blocksExtracted: turnBlocks,
|
|
938
|
+
containsCode: allCodeBlocks.length > 0,
|
|
939
|
+
containsToolCall: hasToolCall
|
|
940
|
+
});
|
|
941
|
+
}
|
|
942
|
+
const artifacts = {
|
|
943
|
+
vaultFiles: [],
|
|
944
|
+
blocksExtracted: allBlocks,
|
|
945
|
+
codeBlocks: allCodeBlocks,
|
|
946
|
+
toolCalls: allToolCalls
|
|
947
|
+
};
|
|
948
|
+
const artifactResults = scenario.artifactChecks.map((check2) => {
|
|
949
|
+
if (config.artifactChecker) {
|
|
950
|
+
const custom = config.artifactChecker(check2, artifacts);
|
|
951
|
+
if (custom) return { check: check2, ...custom };
|
|
952
|
+
}
|
|
953
|
+
switch (check2.type) {
|
|
954
|
+
case "block_extracted": {
|
|
955
|
+
const count = allBlocks.filter((b) => b.type === check2.target).length;
|
|
956
|
+
return {
|
|
957
|
+
check: check2,
|
|
958
|
+
passed: count >= (check2.minCount ?? 1),
|
|
959
|
+
detail: `Found ${count} ${check2.target} blocks (need ${check2.minCount ?? 1})`
|
|
960
|
+
};
|
|
961
|
+
}
|
|
962
|
+
case "code_valid": {
|
|
963
|
+
const hasCode = allCodeBlocks.some(
|
|
964
|
+
(b) => b.language === check2.target || b.code.includes(check2.target)
|
|
965
|
+
);
|
|
966
|
+
return { check: check2, passed: hasCode, detail: hasCode ? "Code block found" : "No matching code" };
|
|
967
|
+
}
|
|
968
|
+
default:
|
|
969
|
+
return {
|
|
970
|
+
check: check2,
|
|
971
|
+
passed: false,
|
|
972
|
+
detail: `Check type "${check2.type}" requires live environment`
|
|
973
|
+
};
|
|
974
|
+
}
|
|
975
|
+
});
|
|
976
|
+
const judgeInput = { scenario, turns, artifacts };
|
|
977
|
+
const judgeResults = [];
|
|
978
|
+
let failedJudges = 0;
|
|
979
|
+
for (const judge of config.judges) {
|
|
980
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
981
|
+
try {
|
|
982
|
+
if (attempt > 0) {
|
|
983
|
+
const wait = attempt * 1e4;
|
|
984
|
+
console.log(` judge retry ${attempt}/2 (waiting ${wait / 1e3}s)`);
|
|
985
|
+
await new Promise((r) => setTimeout(r, wait));
|
|
986
|
+
}
|
|
987
|
+
const scores2 = await judge(tc, judgeInput);
|
|
988
|
+
judgeResults.push(scores2);
|
|
989
|
+
await new Promise((r) => setTimeout(r, 3e3));
|
|
990
|
+
break;
|
|
991
|
+
} catch (err) {
|
|
992
|
+
if (err instanceof JudgeParseError) {
|
|
993
|
+
failedJudges++;
|
|
994
|
+
break;
|
|
774
995
|
}
|
|
996
|
+
if (attempt === 2) failedJudges++;
|
|
775
997
|
}
|
|
776
998
|
}
|
|
777
999
|
}
|
|
@@ -799,7 +1021,7 @@ async function executeScenario(tc, scenario, config) {
|
|
|
799
1021
|
turns,
|
|
800
1022
|
artifactResults,
|
|
801
1023
|
judgeScores: allScores,
|
|
802
|
-
judgeErrors: errorScores.length,
|
|
1024
|
+
judgeErrors: errorScores.length + failedJudges,
|
|
803
1025
|
overallScore,
|
|
804
1026
|
totalDurationMs: Date.now() - startTime,
|
|
805
1027
|
artifacts
|
|
@@ -1451,11 +1673,11 @@ var FileSystemFeedbackTrajectoryStore = class {
|
|
|
1451
1673
|
}
|
|
1452
1674
|
async load() {
|
|
1453
1675
|
if (this.loaded) return;
|
|
1454
|
-
const { readFile } = await import("fs/promises");
|
|
1676
|
+
const { readFile: readFile2 } = await import("fs/promises");
|
|
1455
1677
|
const { join: join4 } = await import("path");
|
|
1456
1678
|
const file = join4(this.dir, "feedback-trajectories.ndjson");
|
|
1457
1679
|
try {
|
|
1458
|
-
const raw = await
|
|
1680
|
+
const raw = await readFile2(file, "utf8");
|
|
1459
1681
|
for (const line of raw.split("\n")) {
|
|
1460
1682
|
if (!line.trim()) continue;
|
|
1461
1683
|
try {
|
|
@@ -1824,380 +2046,169 @@ async function preflightModels(opts) {
|
|
|
1824
2046
|
return { succeeded: true, value: results, error: null };
|
|
1825
2047
|
}
|
|
1826
2048
|
var ModelsUnreachableError = class extends AgentEvalError {
|
|
1827
|
-
constructor(message, results) {
|
|
1828
|
-
super("config", message);
|
|
1829
|
-
this.results = results;
|
|
1830
|
-
this.name = "ModelsUnreachableError";
|
|
1831
|
-
}
|
|
1832
|
-
results;
|
|
1833
|
-
};
|
|
1834
|
-
function describeFailure(r) {
|
|
1835
|
-
if (!r.listed) {
|
|
1836
|
-
const probeNote = r.served === false ? ` (probe ${r.status}${r.detail ? `: ${r.detail}` : ""})` : "";
|
|
1837
|
-
return `${r.model}: not in /models${probeNote}`;
|
|
1838
|
-
}
|
|
1839
|
-
return `${r.model}: listed but probe ${r.status}${r.detail ? ` \u2014 ${r.detail}` : ""}`;
|
|
1840
|
-
}
|
|
1841
|
-
async function assertModelsServed(opts) {
|
|
1842
|
-
const outcome = await preflightModels(opts);
|
|
1843
|
-
if (!outcome.succeeded || outcome.value === null) {
|
|
1844
|
-
throw new ConfigError(
|
|
1845
|
-
outcome.error ?? "assertModelsServed: preflight failed without an error message"
|
|
1846
|
-
);
|
|
1847
|
-
}
|
|
1848
|
-
const dead = outcome.value.filter((r) => !r.listed || r.served === false);
|
|
1849
|
-
if (dead.length > 0) {
|
|
1850
|
-
throw new ModelsUnreachableError(
|
|
1851
|
-
`assertModelsServed: ${dead.length}/${outcome.value.length} model(s) unreachable on the router \u2014 ${dead.map(describeFailure).join("; ")}`,
|
|
1852
|
-
outcome.value
|
|
1853
|
-
);
|
|
1854
|
-
}
|
|
1855
|
-
return outcome.value;
|
|
1856
|
-
}
|
|
1857
|
-
|
|
1858
|
-
// src/integrity/single-backend.ts
|
|
1859
|
-
var SingleBackendError = class extends AgentEvalError {
|
|
1860
|
-
constructor(message, report) {
|
|
1861
|
-
super("backend_integrity", message);
|
|
1862
|
-
this.report = report;
|
|
1863
|
-
this.name = "SingleBackendError";
|
|
1864
|
-
}
|
|
1865
|
-
report;
|
|
1866
|
-
};
|
|
1867
|
-
function stripSlash2(url) {
|
|
1868
|
-
return url.replace(/\/+$/, "");
|
|
1869
|
-
}
|
|
1870
|
-
function assertSingleBackend(agent, judge, opts = {}) {
|
|
1871
|
-
const divergences = [];
|
|
1872
|
-
if (agent.kind !== judge.kind) {
|
|
1873
|
-
divergences.push({ field: "kind", agent: agent.kind, judge: judge.kind });
|
|
1874
|
-
}
|
|
1875
|
-
if (stripSlash2(agent.baseUrl) !== stripSlash2(judge.baseUrl)) {
|
|
1876
|
-
divergences.push({ field: "baseUrl", agent: agent.baseUrl, judge: judge.baseUrl });
|
|
1877
|
-
}
|
|
1878
|
-
if (agent.model !== judge.model) {
|
|
1879
|
-
divergences.push({ field: "model", agent: agent.model, judge: judge.model });
|
|
1880
|
-
}
|
|
1881
|
-
if (agent.provider !== judge.provider) {
|
|
1882
|
-
divergences.push({ field: "provider", agent: agent.provider, judge: judge.provider });
|
|
1883
|
-
}
|
|
1884
|
-
const agentHasKey = Boolean(agent.apiKey);
|
|
1885
|
-
const judgeHasKey = Boolean(judge.apiKey);
|
|
1886
|
-
if (agentHasKey !== judgeHasKey) {
|
|
1887
|
-
divergences.push({
|
|
1888
|
-
field: "apiKeyPresence",
|
|
1889
|
-
agent: agentHasKey ? "set" : "empty",
|
|
1890
|
-
judge: judgeHasKey ? "set" : "empty"
|
|
1891
|
-
});
|
|
1892
|
-
}
|
|
1893
|
-
const blocking = opts.strict ? divergences : divergences.filter((d) => d.field !== "model");
|
|
1894
|
-
const ok = blocking.length === 0;
|
|
1895
|
-
const report = { ok, divergences };
|
|
1896
|
-
if (!ok) {
|
|
1897
|
-
const agentLabel = opts.agentLabel ?? "agent";
|
|
1898
|
-
const judgeLabel = opts.judgeLabel ?? "judge";
|
|
1899
|
-
const detail = blocking.map((d) => `${d.field}: ${agentLabel}=${d.agent ?? "\u2205"} vs ${judgeLabel}=${d.judge ?? "\u2205"}`).join("; ");
|
|
1900
|
-
throw new SingleBackendError(
|
|
1901
|
-
`single-backend: ${agentLabel} and ${judgeLabel} backends diverge \u2014 the judge would re-route through a different backend than the agent (${detail})`,
|
|
1902
|
-
report
|
|
1903
|
-
);
|
|
1904
|
-
}
|
|
1905
|
-
return report;
|
|
1906
|
-
}
|
|
1907
|
-
|
|
1908
|
-
// src/judge-families.ts
|
|
1909
|
-
var PROVIDER_PREFIX = {
|
|
1910
|
-
anthropic: "anthropic",
|
|
1911
|
-
openai: "openai",
|
|
1912
|
-
"azure-openai": "openai",
|
|
1913
|
-
google: "google",
|
|
1914
|
-
"google-vertex": "google",
|
|
1915
|
-
meta: "meta",
|
|
1916
|
-
"meta-llama": "meta",
|
|
1917
|
-
mistral: "mistral",
|
|
1918
|
-
mistralai: "mistral",
|
|
1919
|
-
deepseek: "deepseek",
|
|
1920
|
-
xai: "xai",
|
|
1921
|
-
qwen: "qwen",
|
|
1922
|
-
alibaba: "qwen",
|
|
1923
|
-
cohere: "cohere",
|
|
1924
|
-
amazon: "amazon",
|
|
1925
|
-
bedrock: "amazon",
|
|
1926
|
-
moonshot: "moonshot",
|
|
1927
|
-
moonshotai: "moonshot",
|
|
1928
|
-
kimi: "moonshot",
|
|
1929
|
-
"kimi-code": "moonshot",
|
|
1930
|
-
zhipu: "zhipu",
|
|
1931
|
-
zhipuai: "zhipu",
|
|
1932
|
-
zai: "zhipu",
|
|
1933
|
-
"z-ai": "zhipu",
|
|
1934
|
-
glm: "zhipu"
|
|
1935
|
-
};
|
|
1936
|
-
var NAME_PATTERNS = [
|
|
1937
|
-
[/claude/i, "anthropic"],
|
|
1938
|
-
[/\b(gpt|davinci|babbage)\b|^o[134]\b|[-/]o[134]\b|gpt-/i, "openai"],
|
|
1939
|
-
[/gemini|palm|gemma|bison/i, "google"],
|
|
1940
|
-
[/llama/i, "meta"],
|
|
1941
|
-
[/mi(s|x)tral|codestral|magistral/i, "mistral"],
|
|
1942
|
-
[/deepseek/i, "deepseek"],
|
|
1943
|
-
[/grok/i, "xai"],
|
|
1944
|
-
[/qwen/i, "qwen"],
|
|
1945
|
-
[/command-?(r|a)?/i, "cohere"],
|
|
1946
|
-
[/\b(nova|titan)\b/i, "amazon"],
|
|
1947
|
-
[/\bkimi\b|moonshot/i, "moonshot"],
|
|
1948
|
-
[/\bglm\b|zhipu|\bz-?ai\b/i, "zhipu"]
|
|
1949
|
-
];
|
|
1950
|
-
function judgeFamily(modelId) {
|
|
1951
|
-
const id = modelId.trim().split("@")[0].toLowerCase();
|
|
1952
|
-
const slash = id.indexOf("/");
|
|
1953
|
-
if (slash > 0) {
|
|
1954
|
-
const prefix = id.slice(0, slash);
|
|
1955
|
-
const mapped = PROVIDER_PREFIX[prefix];
|
|
1956
|
-
if (mapped) return mapped;
|
|
1957
|
-
}
|
|
1958
|
-
for (const [pattern, family] of NAME_PATTERNS) {
|
|
1959
|
-
if (pattern.test(id)) return family;
|
|
1960
|
-
}
|
|
1961
|
-
return "unknown";
|
|
1962
|
-
}
|
|
1963
|
-
var CrossFamilyError = class extends Error {
|
|
1964
|
-
constructor(message, families, models) {
|
|
1965
|
-
super(message);
|
|
1966
|
-
this.families = families;
|
|
1967
|
-
this.models = models;
|
|
1968
|
-
this.name = "CrossFamilyError";
|
|
1969
|
-
}
|
|
1970
|
-
families;
|
|
1971
|
-
models;
|
|
1972
|
-
};
|
|
1973
|
-
function assertCrossFamily(models, opts = {}) {
|
|
1974
|
-
const minFamilies = opts.minFamilies ?? 2;
|
|
1975
|
-
const families = /* @__PURE__ */ new Set();
|
|
1976
|
-
for (const m of models) {
|
|
1977
|
-
const f = judgeFamily(m);
|
|
1978
|
-
if (f === "unknown" && !opts.allowUnknown) continue;
|
|
1979
|
-
families.add(f);
|
|
1980
|
-
}
|
|
1981
|
-
const list = [...families].sort();
|
|
1982
|
-
if (list.length < minFamilies) {
|
|
1983
|
-
throw new CrossFamilyError(
|
|
1984
|
-
`judge ensemble spans ${list.length} provider famil${list.length === 1 ? "y" : "ies"} (${list.join(", ") || "none"}) but ${minFamilies} required \u2014 a single-family ensemble is correlated bias, not independent signal`,
|
|
1985
|
-
list,
|
|
1986
|
-
models
|
|
1987
|
-
);
|
|
1988
|
-
}
|
|
1989
|
-
return list;
|
|
1990
|
-
}
|
|
1991
|
-
|
|
1992
|
-
// src/judges.ts
|
|
1993
|
-
function createDomainExpertJudge(domain) {
|
|
1994
|
-
return async (tc, { scenario, turns }) => {
|
|
1995
|
-
const conversation = turns.map(
|
|
1996
|
-
(t, i) => `Turn ${i + 1}:
|
|
1997
|
-
User: ${t.userMessage}
|
|
1998
|
-
Agent: ${t.agentResponse.slice(0, 2e3)}`
|
|
1999
|
-
).join("\n\n---\n\n");
|
|
2000
|
-
const resp = await tc.chat({
|
|
2001
|
-
model: "gpt-4o",
|
|
2002
|
-
messages: [
|
|
2003
|
-
{
|
|
2004
|
-
role: "system",
|
|
2005
|
-
content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
|
|
2006
|
-
|
|
2007
|
-
Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
|
|
2008
|
-
|
|
2009
|
-
Evaluate:
|
|
2010
|
-
1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
|
|
2011
|
-
2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
|
|
2012
|
-
|
|
2013
|
-
Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
|
|
2014
|
-
},
|
|
2015
|
-
{
|
|
2016
|
-
role: "user",
|
|
2017
|
-
content: `Persona: ${scenario.persona} (${scenario.label})
|
|
2018
|
-
Scenario: ${scenario.thesis}
|
|
2019
|
-
|
|
2020
|
-
${conversation}`
|
|
2021
|
-
}
|
|
2022
|
-
],
|
|
2023
|
-
temperature: 0.1,
|
|
2024
|
-
maxTokens: 800
|
|
2025
|
-
});
|
|
2026
|
-
return parseJudgeResponse("domain_expert", resp);
|
|
2027
|
-
};
|
|
2028
|
-
}
|
|
2029
|
-
var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
|
|
2030
|
-
const codeBlocks = artifacts.codeBlocks;
|
|
2031
|
-
if (codeBlocks.length === 0) {
|
|
2032
|
-
return [
|
|
2033
|
-
{
|
|
2034
|
-
judgeName: "code_execution",
|
|
2035
|
-
dimension: "code_execution",
|
|
2036
|
-
score: 0,
|
|
2037
|
-
reasoning: "No code blocks found in agent response."
|
|
2038
|
-
}
|
|
2039
|
-
];
|
|
2040
|
-
}
|
|
2041
|
-
const codeText = codeBlocks.map(
|
|
2042
|
-
(b, i) => `Block ${i + 1} (${b.language}):
|
|
2043
|
-
\`\`\`${b.language}
|
|
2044
|
-
${b.code.slice(0, 3e3)}
|
|
2045
|
-
\`\`\``
|
|
2046
|
-
).join("\n\n");
|
|
2047
|
-
const resp = await tc.chat({
|
|
2048
|
-
model: "gpt-4o",
|
|
2049
|
-
messages: [
|
|
2050
|
-
{
|
|
2051
|
-
role: "system",
|
|
2052
|
-
content: `You are a principal software engineer reviewing code written by an AI agent.
|
|
2053
|
-
|
|
2054
|
-
Score STRICTLY:
|
|
2055
|
-
1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
|
|
2056
|
-
2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
|
|
2057
|
-
3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
|
|
2058
|
-
|
|
2059
|
-
Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
|
|
2060
|
-
},
|
|
2061
|
-
{
|
|
2062
|
-
role: "user",
|
|
2063
|
-
content: `Task: ${scenario.thesis}
|
|
2064
|
-
|
|
2065
|
-
${codeText}`
|
|
2066
|
-
}
|
|
2067
|
-
],
|
|
2068
|
-
temperature: 0.1,
|
|
2069
|
-
maxTokens: 1e3
|
|
2070
|
-
});
|
|
2071
|
-
return parseJudgeResponse("code_execution", resp);
|
|
2072
|
-
};
|
|
2073
|
-
var coherenceJudge = async (tc, { scenario, turns }) => {
|
|
2074
|
-
if (turns.length < 2) {
|
|
2075
|
-
return [];
|
|
2076
|
-
}
|
|
2077
|
-
const conversation = turns.map(
|
|
2078
|
-
(t, i) => `Turn ${i + 1}:
|
|
2079
|
-
User: ${t.userMessage}
|
|
2080
|
-
Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
|
|
2081
|
-
).join("\n\n---\n\n");
|
|
2082
|
-
const resp = await tc.chat({
|
|
2083
|
-
model: "gpt-4o",
|
|
2084
|
-
messages: [
|
|
2085
|
-
{
|
|
2086
|
-
role: "system",
|
|
2087
|
-
content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
|
|
2088
|
-
|
|
2089
|
-
Score STRICTLY:
|
|
2090
|
-
1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
|
|
2091
|
-
2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
|
|
2092
|
-
3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
|
|
2093
|
-
|
|
2094
|
-
Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
|
|
2095
|
-
},
|
|
2096
|
-
{
|
|
2097
|
-
role: "user",
|
|
2098
|
-
content: `Scenario: ${scenario.thesis}
|
|
2099
|
-
|
|
2100
|
-
${conversation}`
|
|
2101
|
-
}
|
|
2102
|
-
],
|
|
2103
|
-
temperature: 0.1,
|
|
2104
|
-
maxTokens: 800
|
|
2105
|
-
});
|
|
2106
|
-
return parseJudgeResponse("coherence", resp);
|
|
2107
|
-
};
|
|
2108
|
-
var adversarialJudge = async (tc, { scenario, turns }) => {
|
|
2109
|
-
const conversation = turns.map(
|
|
2110
|
-
(t, i) => `Turn ${i + 1}:
|
|
2111
|
-
User: ${t.userMessage}
|
|
2112
|
-
Agent: ${t.agentResponse.slice(0, 1500)}`
|
|
2113
|
-
).join("\n\n---\n\n");
|
|
2114
|
-
const resp = await tc.chat({
|
|
2115
|
-
model: "gpt-4o",
|
|
2116
|
-
messages: [
|
|
2117
|
-
{
|
|
2118
|
-
role: "system",
|
|
2119
|
-
content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
|
|
2120
|
-
|
|
2121
|
-
1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
|
|
2122
|
-
2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
|
|
2123
|
-
3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
|
|
2124
|
-
|
|
2125
|
-
Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
|
|
2126
|
-
|
|
2127
|
-
Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
|
|
2128
|
-
},
|
|
2129
|
-
{
|
|
2130
|
-
role: "user",
|
|
2131
|
-
content: `Persona: ${scenario.persona}
|
|
2132
|
-
Scenario: ${scenario.thesis}
|
|
2133
|
-
|
|
2134
|
-
${conversation}`
|
|
2135
|
-
}
|
|
2136
|
-
],
|
|
2137
|
-
temperature: 0.2,
|
|
2138
|
-
maxTokens: 800
|
|
2139
|
-
});
|
|
2140
|
-
return parseJudgeResponse("adversarial", resp);
|
|
2049
|
+
constructor(message, results) {
|
|
2050
|
+
super("config", message);
|
|
2051
|
+
this.results = results;
|
|
2052
|
+
this.name = "ModelsUnreachableError";
|
|
2053
|
+
}
|
|
2054
|
+
results;
|
|
2141
2055
|
};
|
|
2142
|
-
function
|
|
2143
|
-
|
|
2144
|
-
const
|
|
2145
|
-
|
|
2146
|
-
|
|
2147
|
-
|
|
2148
|
-
|
|
2149
|
-
|
|
2150
|
-
|
|
2151
|
-
|
|
2152
|
-
|
|
2153
|
-
|
|
2154
|
-
|
|
2155
|
-
|
|
2156
|
-
|
|
2157
|
-
|
|
2158
|
-
|
|
2159
|
-
|
|
2056
|
+
function describeFailure(r) {
|
|
2057
|
+
if (!r.listed) {
|
|
2058
|
+
const probeNote = r.served === false ? ` (probe ${r.status}${r.detail ? `: ${r.detail}` : ""})` : "";
|
|
2059
|
+
return `${r.model}: not in /models${probeNote}`;
|
|
2060
|
+
}
|
|
2061
|
+
return `${r.model}: listed but probe ${r.status}${r.detail ? ` \u2014 ${r.detail}` : ""}`;
|
|
2062
|
+
}
|
|
2063
|
+
async function assertModelsServed(opts) {
|
|
2064
|
+
const outcome = await preflightModels(opts);
|
|
2065
|
+
if (!outcome.succeeded || outcome.value === null) {
|
|
2066
|
+
throw new ConfigError(
|
|
2067
|
+
outcome.error ?? "assertModelsServed: preflight failed without an error message"
|
|
2068
|
+
);
|
|
2069
|
+
}
|
|
2070
|
+
const dead = outcome.value.filter((r) => !r.listed || r.served === false);
|
|
2071
|
+
if (dead.length > 0) {
|
|
2072
|
+
throw new ModelsUnreachableError(
|
|
2073
|
+
`assertModelsServed: ${dead.length}/${outcome.value.length} model(s) unreachable on the router \u2014 ${dead.map(describeFailure).join("; ")}`,
|
|
2074
|
+
outcome.value
|
|
2075
|
+
);
|
|
2076
|
+
}
|
|
2077
|
+
return outcome.value;
|
|
2078
|
+
}
|
|
2160
2079
|
|
|
2161
|
-
|
|
2162
|
-
|
|
2163
|
-
|
|
2164
|
-
|
|
2165
|
-
|
|
2080
|
+
// src/integrity/single-backend.ts
|
|
2081
|
+
var SingleBackendError = class extends AgentEvalError {
|
|
2082
|
+
constructor(message, report) {
|
|
2083
|
+
super("backend_integrity", message);
|
|
2084
|
+
this.report = report;
|
|
2085
|
+
this.name = "SingleBackendError";
|
|
2086
|
+
}
|
|
2087
|
+
report;
|
|
2088
|
+
};
|
|
2089
|
+
function stripSlash2(url) {
|
|
2090
|
+
return url.replace(/\/+$/, "");
|
|
2091
|
+
}
|
|
2092
|
+
function assertSingleBackend(agent, judge, opts = {}) {
|
|
2093
|
+
const divergences = [];
|
|
2094
|
+
if (agent.kind !== judge.kind) {
|
|
2095
|
+
divergences.push({ field: "kind", agent: agent.kind, judge: judge.kind });
|
|
2096
|
+
}
|
|
2097
|
+
if (stripSlash2(agent.baseUrl) !== stripSlash2(judge.baseUrl)) {
|
|
2098
|
+
divergences.push({ field: "baseUrl", agent: agent.baseUrl, judge: judge.baseUrl });
|
|
2099
|
+
}
|
|
2100
|
+
if (agent.model !== judge.model) {
|
|
2101
|
+
divergences.push({ field: "model", agent: agent.model, judge: judge.model });
|
|
2102
|
+
}
|
|
2103
|
+
if (agent.provider !== judge.provider) {
|
|
2104
|
+
divergences.push({ field: "provider", agent: agent.provider, judge: judge.provider });
|
|
2105
|
+
}
|
|
2106
|
+
const agentHasKey = Boolean(agent.apiKey);
|
|
2107
|
+
const judgeHasKey = Boolean(judge.apiKey);
|
|
2108
|
+
if (agentHasKey !== judgeHasKey) {
|
|
2109
|
+
divergences.push({
|
|
2110
|
+
field: "apiKeyPresence",
|
|
2111
|
+
agent: agentHasKey ? "set" : "empty",
|
|
2112
|
+
judge: judgeHasKey ? "set" : "empty"
|
|
2166
2113
|
});
|
|
2167
|
-
|
|
2168
|
-
|
|
2114
|
+
}
|
|
2115
|
+
const blocking = opts.strict ? divergences : divergences.filter((d) => d.field !== "model");
|
|
2116
|
+
const ok = blocking.length === 0;
|
|
2117
|
+
const report = { ok, divergences };
|
|
2118
|
+
if (!ok) {
|
|
2119
|
+
const agentLabel = opts.agentLabel ?? "agent";
|
|
2120
|
+
const judgeLabel = opts.judgeLabel ?? "judge";
|
|
2121
|
+
const detail = blocking.map((d) => `${d.field}: ${agentLabel}=${d.agent ?? "\u2205"} vs ${judgeLabel}=${d.judge ?? "\u2205"}`).join("; ");
|
|
2122
|
+
throw new SingleBackendError(
|
|
2123
|
+
`single-backend: ${agentLabel} and ${judgeLabel} backends diverge \u2014 the judge would re-route through a different backend than the agent (${detail})`,
|
|
2124
|
+
report
|
|
2125
|
+
);
|
|
2126
|
+
}
|
|
2127
|
+
return report;
|
|
2169
2128
|
}
|
|
2170
|
-
|
|
2171
|
-
|
|
2129
|
+
|
|
2130
|
+
// src/judge-families.ts
|
|
2131
|
+
var PROVIDER_PREFIX = {
|
|
2132
|
+
anthropic: "anthropic",
|
|
2133
|
+
openai: "openai",
|
|
2134
|
+
"azure-openai": "openai",
|
|
2135
|
+
google: "google",
|
|
2136
|
+
"google-vertex": "google",
|
|
2137
|
+
meta: "meta",
|
|
2138
|
+
"meta-llama": "meta",
|
|
2139
|
+
mistral: "mistral",
|
|
2140
|
+
mistralai: "mistral",
|
|
2141
|
+
deepseek: "deepseek",
|
|
2142
|
+
xai: "xai",
|
|
2143
|
+
qwen: "qwen",
|
|
2144
|
+
alibaba: "qwen",
|
|
2145
|
+
cohere: "cohere",
|
|
2146
|
+
amazon: "amazon",
|
|
2147
|
+
bedrock: "amazon",
|
|
2148
|
+
moonshot: "moonshot",
|
|
2149
|
+
moonshotai: "moonshot",
|
|
2150
|
+
kimi: "moonshot",
|
|
2151
|
+
"kimi-code": "moonshot",
|
|
2152
|
+
zhipu: "zhipu",
|
|
2153
|
+
zhipuai: "zhipu",
|
|
2154
|
+
zai: "zhipu",
|
|
2155
|
+
"z-ai": "zhipu",
|
|
2156
|
+
glm: "zhipu"
|
|
2157
|
+
};
|
|
2158
|
+
var NAME_PATTERNS = [
|
|
2159
|
+
[/claude/i, "anthropic"],
|
|
2160
|
+
[/\b(gpt|davinci|babbage)\b|^o[134]\b|[-/]o[134]\b|gpt-/i, "openai"],
|
|
2161
|
+
[/gemini|palm|gemma|bison/i, "google"],
|
|
2162
|
+
[/llama/i, "meta"],
|
|
2163
|
+
[/mi(s|x)tral|codestral|magistral/i, "mistral"],
|
|
2164
|
+
[/deepseek/i, "deepseek"],
|
|
2165
|
+
[/grok/i, "xai"],
|
|
2166
|
+
[/qwen/i, "qwen"],
|
|
2167
|
+
[/command-?(r|a)?/i, "cohere"],
|
|
2168
|
+
[/\b(nova|titan)\b/i, "amazon"],
|
|
2169
|
+
[/\bkimi\b|moonshot/i, "moonshot"],
|
|
2170
|
+
[/\bglm\b|zhipu|\bz-?ai\b/i, "zhipu"]
|
|
2171
|
+
];
|
|
2172
|
+
function judgeFamily(modelId) {
|
|
2173
|
+
const id = modelId.trim().split("@")[0].toLowerCase();
|
|
2174
|
+
const slash = id.indexOf("/");
|
|
2175
|
+
if (slash > 0) {
|
|
2176
|
+
const prefix = id.slice(0, slash);
|
|
2177
|
+
const mapped = PROVIDER_PREFIX[prefix];
|
|
2178
|
+
if (mapped) return mapped;
|
|
2179
|
+
}
|
|
2180
|
+
for (const [pattern, family] of NAME_PATTERNS) {
|
|
2181
|
+
if (pattern.test(id)) return family;
|
|
2182
|
+
}
|
|
2183
|
+
return "unknown";
|
|
2172
2184
|
}
|
|
2173
|
-
|
|
2174
|
-
|
|
2175
|
-
|
|
2176
|
-
|
|
2177
|
-
|
|
2178
|
-
|
|
2179
|
-
|
|
2180
|
-
|
|
2181
|
-
|
|
2182
|
-
|
|
2183
|
-
|
|
2184
|
-
|
|
2185
|
-
|
|
2186
|
-
|
|
2187
|
-
|
|
2188
|
-
|
|
2189
|
-
|
|
2190
|
-
|
|
2185
|
+
var CrossFamilyError = class extends Error {
|
|
2186
|
+
constructor(message, families, models) {
|
|
2187
|
+
super(message);
|
|
2188
|
+
this.families = families;
|
|
2189
|
+
this.models = models;
|
|
2190
|
+
this.name = "CrossFamilyError";
|
|
2191
|
+
}
|
|
2192
|
+
families;
|
|
2193
|
+
models;
|
|
2194
|
+
};
|
|
2195
|
+
function assertCrossFamily(models, opts = {}) {
|
|
2196
|
+
const minFamilies = opts.minFamilies ?? 2;
|
|
2197
|
+
const families = /* @__PURE__ */ new Set();
|
|
2198
|
+
for (const m of models) {
|
|
2199
|
+
const f = judgeFamily(m);
|
|
2200
|
+
if (f === "unknown" && !opts.allowUnknown) continue;
|
|
2201
|
+
families.add(f);
|
|
2202
|
+
}
|
|
2203
|
+
const list = [...families].sort();
|
|
2204
|
+
if (list.length < minFamilies) {
|
|
2205
|
+
throw new CrossFamilyError(
|
|
2206
|
+
`judge ensemble spans ${list.length} provider famil${list.length === 1 ? "y" : "ies"} (${list.join(", ") || "none"}) but ${minFamilies} required \u2014 a single-family ensemble is correlated bias, not independent signal`,
|
|
2207
|
+
list,
|
|
2208
|
+
models
|
|
2191
2209
|
);
|
|
2192
|
-
return [
|
|
2193
|
-
{
|
|
2194
|
-
judgeName,
|
|
2195
|
-
dimension: "parse_error",
|
|
2196
|
-
score: 0,
|
|
2197
|
-
reasoning: `Parse failed: ${err.message?.slice(0, 100)}. Raw: ${content.slice(0, 200)}`
|
|
2198
|
-
}
|
|
2199
|
-
];
|
|
2200
2210
|
}
|
|
2211
|
+
return list;
|
|
2201
2212
|
}
|
|
2202
2213
|
|
|
2203
2214
|
// src/live-proof.ts
|
|
@@ -2994,17 +3005,181 @@ var DualAgentBench = class {
|
|
|
2994
3005
|
finalScore: lastScore
|
|
2995
3006
|
});
|
|
2996
3007
|
}
|
|
2997
|
-
const convergedResults = results.filter((r) => r.converged);
|
|
2998
|
-
const convergenceRate = results.length ? convergedResults.length / results.length : 0;
|
|
2999
|
-
const avgRoundsToConverge = convergedResults.length ? convergedResults.reduce((acc, r) => acc + (r.roundsToConverge ?? 0), 0) / convergedResults.length : null;
|
|
3000
|
-
const avgFinalScore = results.length ? results.reduce((acc, r) => acc + r.finalScore, 0) / results.length : 0;
|
|
3001
|
-
return {
|
|
3002
|
-
scenarios: results,
|
|
3003
|
-
aggregate: { convergenceRate, avgRoundsToConverge, avgFinalScore },
|
|
3004
|
-
config: { maxRounds, convergenceThreshold: threshold }
|
|
3005
|
-
};
|
|
3008
|
+
const convergedResults = results.filter((r) => r.converged);
|
|
3009
|
+
const convergenceRate = results.length ? convergedResults.length / results.length : 0;
|
|
3010
|
+
const avgRoundsToConverge = convergedResults.length ? convergedResults.reduce((acc, r) => acc + (r.roundsToConverge ?? 0), 0) / convergedResults.length : null;
|
|
3011
|
+
const avgFinalScore = results.length ? results.reduce((acc, r) => acc + r.finalScore, 0) / results.length : 0;
|
|
3012
|
+
return {
|
|
3013
|
+
scenarios: results,
|
|
3014
|
+
aggregate: { convergenceRate, avgRoundsToConverge, avgFinalScore },
|
|
3015
|
+
config: { maxRounds, convergenceThreshold: threshold }
|
|
3016
|
+
};
|
|
3017
|
+
}
|
|
3018
|
+
};
|
|
3019
|
+
|
|
3020
|
+
// src/eval-tools.ts
|
|
3021
|
+
import { readFile } from "fs/promises";
|
|
3022
|
+
function toOpenAiTool(def) {
|
|
3023
|
+
return {
|
|
3024
|
+
type: "function",
|
|
3025
|
+
function: { name: def.name, description: def.description, parameters: def.parameters }
|
|
3026
|
+
};
|
|
3027
|
+
}
|
|
3028
|
+
function makeEvalTools(cfg) {
|
|
3029
|
+
const tools = [];
|
|
3030
|
+
if (cfg.judges) tools.push(runJudgesTool(cfg.judges));
|
|
3031
|
+
if (cfg.completion) tools.push(verifyCompletionTool(cfg.completion.checkCorrectness));
|
|
3032
|
+
if (cfg.analyze) tools.push(analyzeRunsTool(cfg.analyze));
|
|
3033
|
+
return tools;
|
|
3034
|
+
}
|
|
3035
|
+
function runJudgesTool(judges) {
|
|
3036
|
+
if (judges.length === 0) {
|
|
3037
|
+
throw new Error("makeEvalTools: cfg.judges is empty \u2014 supply judges or omit the section");
|
|
3038
|
+
}
|
|
3039
|
+
const names = judges.map((j) => j.name);
|
|
3040
|
+
return {
|
|
3041
|
+
name: "run_judges",
|
|
3042
|
+
description: `Score an artifact with the configured judges (${names.join(", ")}). Pass \`judge\` to run one judge by name; omit it to run all. Returns per-judge scores keyed by judge name.`,
|
|
3043
|
+
parameters: {
|
|
3044
|
+
type: "object",
|
|
3045
|
+
properties: {
|
|
3046
|
+
judge: {
|
|
3047
|
+
type: "string",
|
|
3048
|
+
enum: names,
|
|
3049
|
+
description: "Run only this judge. Omit to run every configured judge."
|
|
3050
|
+
},
|
|
3051
|
+
artifact: {
|
|
3052
|
+
description: "The artifact to score \u2014 any JSON value the judges understand."
|
|
3053
|
+
},
|
|
3054
|
+
scenario: {
|
|
3055
|
+
description: "Optional scenario context forwarded to the judges."
|
|
3056
|
+
}
|
|
3057
|
+
},
|
|
3058
|
+
required: ["artifact"],
|
|
3059
|
+
additionalProperties: false
|
|
3060
|
+
},
|
|
3061
|
+
handler: async (args, ctx) => {
|
|
3062
|
+
const a = requireObjectArgs("run_judges", args);
|
|
3063
|
+
if (!("artifact" in a)) {
|
|
3064
|
+
throw new Error("run_judges: args.artifact is required");
|
|
3065
|
+
}
|
|
3066
|
+
let selected = judges;
|
|
3067
|
+
if (a.judge !== void 0) {
|
|
3068
|
+
if (typeof a.judge !== "string") throw new Error("run_judges: args.judge must be a string");
|
|
3069
|
+
selected = judges.filter((j) => j.name === a.judge);
|
|
3070
|
+
if (selected.length === 0) {
|
|
3071
|
+
throw new Error(
|
|
3072
|
+
`run_judges: unknown judge '${a.judge}' \u2014 configured: ${names.join(", ")}`
|
|
3073
|
+
);
|
|
3074
|
+
}
|
|
3075
|
+
}
|
|
3076
|
+
const signal = ctx?.signal ?? new AbortController().signal;
|
|
3077
|
+
const scenario = a.scenario;
|
|
3078
|
+
const scores2 = {};
|
|
3079
|
+
for (const judge of selected) {
|
|
3080
|
+
if (scenario !== void 0 && judge.appliesTo && !judge.appliesTo(scenario)) continue;
|
|
3081
|
+
scores2[judge.name] = await judge.score({ artifact: a.artifact, scenario, signal });
|
|
3082
|
+
}
|
|
3083
|
+
return { scores: scores2 };
|
|
3084
|
+
}
|
|
3085
|
+
};
|
|
3086
|
+
}
|
|
3087
|
+
function verifyCompletionTool(checkCorrectness) {
|
|
3088
|
+
return {
|
|
3089
|
+
name: "verify_completion",
|
|
3090
|
+
description: "Verify produced state against a gold task spec: each gold requirement is matched to at most one produced artifact/proposal/tool-call, correctness-checked by the host, and reduced to a CompletionVerdict (completionRate, fullyComplete, per-requirement checks).",
|
|
3091
|
+
parameters: {
|
|
3092
|
+
type: "object",
|
|
3093
|
+
properties: {
|
|
3094
|
+
gold: {
|
|
3095
|
+
type: "object",
|
|
3096
|
+
description: "TaskGold \u2014 taskId + requirements the produced state must satisfy."
|
|
3097
|
+
},
|
|
3098
|
+
state: {
|
|
3099
|
+
type: "object",
|
|
3100
|
+
description: "ProducedState \u2014 artifacts, proposals, and tool calls the agent produced."
|
|
3101
|
+
}
|
|
3102
|
+
},
|
|
3103
|
+
required: ["gold", "state"],
|
|
3104
|
+
additionalProperties: false
|
|
3105
|
+
},
|
|
3106
|
+
handler: async (args) => {
|
|
3107
|
+
const a = requireObjectArgs("verify_completion", args);
|
|
3108
|
+
if (typeof a.gold !== "object" || a.gold === null) {
|
|
3109
|
+
throw new Error("verify_completion: args.gold (TaskGold) is required");
|
|
3110
|
+
}
|
|
3111
|
+
if (typeof a.state !== "object" || a.state === null) {
|
|
3112
|
+
throw new Error("verify_completion: args.state (ProducedState) is required");
|
|
3113
|
+
}
|
|
3114
|
+
return verifyCompletion(a.gold, a.state, checkCorrectness);
|
|
3115
|
+
}
|
|
3116
|
+
};
|
|
3117
|
+
}
|
|
3118
|
+
function analyzeRunsTool(analyzeOpts) {
|
|
3119
|
+
return {
|
|
3120
|
+
name: "analyze_runs",
|
|
3121
|
+
description: "Run analyzeRuns over a set of RunRecords and return the InsightReport (composite distribution, per-dimension stats, lift, failure clusters, recommendations). Pass the records inline via `runs`, or `path` to a .json (array) or .jsonl (one record per line) file on the host.",
|
|
3122
|
+
parameters: {
|
|
3123
|
+
type: "object",
|
|
3124
|
+
properties: {
|
|
3125
|
+
runs: {
|
|
3126
|
+
type: "array",
|
|
3127
|
+
items: { type: "object" },
|
|
3128
|
+
description: "RunRecord[] inline."
|
|
3129
|
+
},
|
|
3130
|
+
path: {
|
|
3131
|
+
type: "string",
|
|
3132
|
+
description: "Host path to a JSON array or JSONL file of RunRecords."
|
|
3133
|
+
}
|
|
3134
|
+
},
|
|
3135
|
+
additionalProperties: false
|
|
3136
|
+
},
|
|
3137
|
+
handler: async (args) => {
|
|
3138
|
+
const a = requireObjectArgs("analyze_runs", args);
|
|
3139
|
+
const hasRuns = Array.isArray(a.runs);
|
|
3140
|
+
const hasPath = typeof a.path === "string" && a.path.length > 0;
|
|
3141
|
+
if (hasRuns === hasPath) {
|
|
3142
|
+
throw new Error("analyze_runs: pass exactly one of args.runs (array) or args.path (string)");
|
|
3143
|
+
}
|
|
3144
|
+
const runs = hasRuns ? a.runs : await loadRunRecords(a.path);
|
|
3145
|
+
if (runs.length === 0) {
|
|
3146
|
+
throw new Error("analyze_runs: no runs to analyze");
|
|
3147
|
+
}
|
|
3148
|
+
return analyzeRuns({ ...analyzeOpts, runs });
|
|
3149
|
+
}
|
|
3150
|
+
};
|
|
3151
|
+
}
|
|
3152
|
+
async function loadRunRecords(path) {
|
|
3153
|
+
const text = await readFile(path, "utf8");
|
|
3154
|
+
const trimmed = text.trim();
|
|
3155
|
+
if (trimmed.length === 0) {
|
|
3156
|
+
throw new Error(`analyze_runs: file '${path}' is empty`);
|
|
3157
|
+
}
|
|
3158
|
+
if (trimmed.startsWith("[")) {
|
|
3159
|
+
const parsed = JSON.parse(trimmed);
|
|
3160
|
+
if (!Array.isArray(parsed)) {
|
|
3161
|
+
throw new Error(`analyze_runs: file '${path}' did not parse to an array`);
|
|
3162
|
+
}
|
|
3163
|
+
return parsed;
|
|
3006
3164
|
}
|
|
3007
|
-
|
|
3165
|
+
return trimmed.split("\n").flatMap((line, i) => {
|
|
3166
|
+
const l = line.trim();
|
|
3167
|
+
if (l.length === 0) return [];
|
|
3168
|
+
try {
|
|
3169
|
+
return [JSON.parse(l)];
|
|
3170
|
+
} catch (err) {
|
|
3171
|
+
throw new Error(
|
|
3172
|
+
`analyze_runs: file '${path}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`
|
|
3173
|
+
);
|
|
3174
|
+
}
|
|
3175
|
+
});
|
|
3176
|
+
}
|
|
3177
|
+
function requireObjectArgs(tool, args) {
|
|
3178
|
+
if (typeof args !== "object" || args === null || Array.isArray(args)) {
|
|
3179
|
+
throw new Error(`${tool}: args must be an object`);
|
|
3180
|
+
}
|
|
3181
|
+
return args;
|
|
3182
|
+
}
|
|
3008
3183
|
|
|
3009
3184
|
// src/harness-optimizer.ts
|
|
3010
3185
|
var DEFAULT_HARNESS_OBJECTIVES = [
|
|
@@ -3127,15 +3302,22 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3127
3302
|
for (const d of dimensionKeys) dimAcc[d] = [];
|
|
3128
3303
|
let rationale = "";
|
|
3129
3304
|
let costUsd = 0;
|
|
3305
|
+
const seenCount = /* @__PURE__ */ new Map();
|
|
3306
|
+
const keyFor = (model) => {
|
|
3307
|
+
const n = (seenCount.get(model) ?? 0) + 1;
|
|
3308
|
+
seenCount.set(model, n);
|
|
3309
|
+
return n === 1 ? model : `${model}#${n}`;
|
|
3310
|
+
};
|
|
3130
3311
|
for (const v of verdicts) {
|
|
3131
3312
|
costUsd += v.costUsd ?? 0;
|
|
3313
|
+
const key = keyFor(v.model);
|
|
3132
3314
|
if (!v.perDimension) {
|
|
3133
|
-
failedJudges.push(
|
|
3315
|
+
failedJudges.push(key);
|
|
3134
3316
|
continue;
|
|
3135
3317
|
}
|
|
3136
3318
|
const dims = {};
|
|
3137
3319
|
for (const d of dimensionKeys) dims[d] = clamp01(Number(v.perDimension[d]));
|
|
3138
|
-
perJudge[
|
|
3320
|
+
perJudge[key] = dims;
|
|
3139
3321
|
for (const d of dimensionKeys) dimAcc[d].push(dims[d]);
|
|
3140
3322
|
if (!rationale && typeof v.rationale === "string" && v.rationale) rationale = v.rationale;
|
|
3141
3323
|
}
|
|
@@ -3170,7 +3352,127 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
|
|
|
3170
3352
|
maxDisagreement,
|
|
3171
3353
|
failedJudges,
|
|
3172
3354
|
costUsd,
|
|
3173
|
-
rationale: rationale || "llm-judge"
|
|
3355
|
+
rationale: rationale || "llm-judge",
|
|
3356
|
+
verdicts: [...verdicts]
|
|
3357
|
+
};
|
|
3358
|
+
}
|
|
3359
|
+
|
|
3360
|
+
// src/judge-retry.ts
|
|
3361
|
+
var DEFAULT_MAX_ATTEMPTS = 3;
|
|
3362
|
+
var DEFAULT_TIMEOUT_MS = 3e5;
|
|
3363
|
+
function sleep(ms) {
|
|
3364
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
3365
|
+
}
|
|
3366
|
+
async function withJudgeRetry(judgeFn, policy = {}) {
|
|
3367
|
+
const maxAttempts = policy.maxAttempts ?? DEFAULT_MAX_ATTEMPTS;
|
|
3368
|
+
const timeoutMs = policy.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
3369
|
+
const backoff = policy.backoffMs ?? backoffMs;
|
|
3370
|
+
const isRetryable = policy.isRetryable ?? isTransientLlmError;
|
|
3371
|
+
const models = policy.models && policy.models.length > 0 ? policy.models : [void 0];
|
|
3372
|
+
let totalAttempts = 0;
|
|
3373
|
+
const attemptErrors = [];
|
|
3374
|
+
let lastError;
|
|
3375
|
+
for (const model of models) {
|
|
3376
|
+
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
3377
|
+
totalAttempts += 1;
|
|
3378
|
+
const controller = new AbortController();
|
|
3379
|
+
const timer = setTimeout(() => controller.abort(new Error("TimeoutError")), timeoutMs);
|
|
3380
|
+
try {
|
|
3381
|
+
const value = await judgeFn(model, controller.signal);
|
|
3382
|
+
clearTimeout(timer);
|
|
3383
|
+
return {
|
|
3384
|
+
value,
|
|
3385
|
+
succeeded: true,
|
|
3386
|
+
attempts: totalAttempts,
|
|
3387
|
+
modelUsed: model,
|
|
3388
|
+
attemptErrors
|
|
3389
|
+
};
|
|
3390
|
+
} catch (err) {
|
|
3391
|
+
clearTimeout(timer);
|
|
3392
|
+
const errObj = err instanceof Error ? err : new Error(String(err));
|
|
3393
|
+
lastError = errObj;
|
|
3394
|
+
attemptErrors.push({
|
|
3395
|
+
attempt: totalAttempts,
|
|
3396
|
+
model: model ?? "(default)",
|
|
3397
|
+
error: errObj.message
|
|
3398
|
+
});
|
|
3399
|
+
if (!isRetryable(errObj)) {
|
|
3400
|
+
return {
|
|
3401
|
+
value: null,
|
|
3402
|
+
succeeded: false,
|
|
3403
|
+
attempts: totalAttempts,
|
|
3404
|
+
error: errObj,
|
|
3405
|
+
attemptErrors
|
|
3406
|
+
};
|
|
3407
|
+
}
|
|
3408
|
+
if (attempt < maxAttempts - 1) {
|
|
3409
|
+
await sleep(backoff(attempt));
|
|
3410
|
+
}
|
|
3411
|
+
}
|
|
3412
|
+
}
|
|
3413
|
+
}
|
|
3414
|
+
return {
|
|
3415
|
+
value: null,
|
|
3416
|
+
succeeded: false,
|
|
3417
|
+
attempts: totalAttempts,
|
|
3418
|
+
error: lastError,
|
|
3419
|
+
attemptErrors
|
|
3420
|
+
};
|
|
3421
|
+
}
|
|
3422
|
+
|
|
3423
|
+
// src/judge-panel.ts
|
|
3424
|
+
function ensembleJudge(opts) {
|
|
3425
|
+
if (opts.models.length === 0) {
|
|
3426
|
+
throw new Error(`ensembleJudge '${opts.name}': models is empty \u2014 nothing to score with`);
|
|
3427
|
+
}
|
|
3428
|
+
if (opts.dimensions.length === 0) {
|
|
3429
|
+
throw new Error(`ensembleJudge '${opts.name}': dimensions is empty \u2014 nothing to score`);
|
|
3430
|
+
}
|
|
3431
|
+
if (opts.crossFamily !== false) {
|
|
3432
|
+
assertCrossFamily(opts.models);
|
|
3433
|
+
}
|
|
3434
|
+
const scoreOne = async (model, input) => {
|
|
3435
|
+
if (opts.retry) {
|
|
3436
|
+
const outcome = await withJudgeRetry((m) => opts.scoreWith(m, input), {
|
|
3437
|
+
...opts.retry,
|
|
3438
|
+
models: [model]
|
|
3439
|
+
});
|
|
3440
|
+
if (!outcome.succeeded || outcome.value === null) {
|
|
3441
|
+
return {
|
|
3442
|
+
model,
|
|
3443
|
+
perDimension: null,
|
|
3444
|
+
rationale: outcome.error?.message ?? "judge failed after retries"
|
|
3445
|
+
};
|
|
3446
|
+
}
|
|
3447
|
+
return outcome.value;
|
|
3448
|
+
}
|
|
3449
|
+
try {
|
|
3450
|
+
return await opts.scoreWith(model, input);
|
|
3451
|
+
} catch (err) {
|
|
3452
|
+
return {
|
|
3453
|
+
model,
|
|
3454
|
+
perDimension: null,
|
|
3455
|
+
rationale: err instanceof Error ? err.message : String(err)
|
|
3456
|
+
};
|
|
3457
|
+
}
|
|
3458
|
+
};
|
|
3459
|
+
return {
|
|
3460
|
+
name: opts.name,
|
|
3461
|
+
dimensions: opts.dimensions.map((d) => ({ key: d, description: d })),
|
|
3462
|
+
async score({ artifact, scenario }) {
|
|
3463
|
+
const input = { artifact, scenario };
|
|
3464
|
+
const verdicts = await Promise.all(opts.models.map((model) => scoreOne(model, input)));
|
|
3465
|
+
const agg = aggregateJudgeVerdicts(verdicts, opts.dimensions, opts.weights);
|
|
3466
|
+
const score = {
|
|
3467
|
+
dimensions: agg.perDimension,
|
|
3468
|
+
composite: agg.composite,
|
|
3469
|
+
notes: agg.rationale,
|
|
3470
|
+
maxDisagreement: agg.maxDisagreement,
|
|
3471
|
+
perJudge: agg.perJudge
|
|
3472
|
+
};
|
|
3473
|
+
if (agg.failedJudges.length > 0) score.failedJudges = agg.failedJudges;
|
|
3474
|
+
return score;
|
|
3475
|
+
}
|
|
3174
3476
|
};
|
|
3175
3477
|
}
|
|
3176
3478
|
|
|
@@ -3505,7 +3807,7 @@ function createAxService(provider, apiKey, model) {
|
|
|
3505
3807
|
});
|
|
3506
3808
|
}
|
|
3507
3809
|
function seededShuffle(items, seed) {
|
|
3508
|
-
const rng =
|
|
3810
|
+
const rng = mulberry322(hashString(seed));
|
|
3509
3811
|
const out = [...items];
|
|
3510
3812
|
for (let i = out.length - 1; i > 0; i--) {
|
|
3511
3813
|
const j = Math.floor(rng() * (i + 1));
|
|
@@ -3521,7 +3823,7 @@ function hashString(value) {
|
|
|
3521
3823
|
}
|
|
3522
3824
|
return h >>> 0;
|
|
3523
3825
|
}
|
|
3524
|
-
function
|
|
3826
|
+
function mulberry322(seed) {
|
|
3525
3827
|
let a = seed >>> 0;
|
|
3526
3828
|
return () => {
|
|
3527
3829
|
a = a + 1831565813 | 0;
|
|
@@ -4856,42 +5158,6 @@ function formatScorecardDiff(diff) {
|
|
|
4856
5158
|
return lines.join("\n");
|
|
4857
5159
|
}
|
|
4858
5160
|
|
|
4859
|
-
// src/series-convergence.ts
|
|
4860
|
-
function analyzeSeries(values, options = {}) {
|
|
4861
|
-
const window = options.window ?? 5;
|
|
4862
|
-
const stableCv = options.stableCv ?? 0.05;
|
|
4863
|
-
const driftRun = options.driftRun ?? 3;
|
|
4864
|
-
if (values.length < Math.max(2, Math.min(window, 3))) {
|
|
4865
|
-
return { state: "insufficient-data", windowMean: 0, windowCv: 0, tailRun: 0, stable: false };
|
|
4866
|
-
}
|
|
4867
|
-
const tail = values.slice(-window);
|
|
4868
|
-
const mean5 = tail.reduce((a, b) => a + b, 0) / tail.length;
|
|
4869
|
-
const variance = tail.reduce((acc, v) => acc + (v - mean5) ** 2, 0) / tail.length;
|
|
4870
|
-
const stdDev = Math.sqrt(variance);
|
|
4871
|
-
const refMean = Math.abs(mean5) > 1e-9 ? Math.abs(mean5) : 1;
|
|
4872
|
-
const cv = stdDev / refMean;
|
|
4873
|
-
const stable = tail.length >= window && cv <= stableCv;
|
|
4874
|
-
let tailRun = 0;
|
|
4875
|
-
let direction = 0;
|
|
4876
|
-
for (let i = values.length - 1; i > 0; i--) {
|
|
4877
|
-
const delta = values[i] - values[i - 1];
|
|
4878
|
-
if (delta === 0) break;
|
|
4879
|
-
const dir = delta > 0 ? 1 : -1;
|
|
4880
|
-
if (direction === 0) direction = dir;
|
|
4881
|
-
if (dir !== direction) break;
|
|
4882
|
-
tailRun += dir;
|
|
4883
|
-
}
|
|
4884
|
-
let state;
|
|
4885
|
-
if (stable) {
|
|
4886
|
-
state = "stabilized";
|
|
4887
|
-
} else if (Math.abs(tailRun) >= driftRun) {
|
|
4888
|
-
state = tailRun > 0 ? "drifting-up" : "drifting-down";
|
|
4889
|
-
} else {
|
|
4890
|
-
state = "noisy";
|
|
4891
|
-
}
|
|
4892
|
-
return { state, windowMean: mean5, windowCv: cv, tailRun, stable };
|
|
4893
|
-
}
|
|
4894
|
-
|
|
4895
5161
|
// src/slo.ts
|
|
4896
5162
|
function checkSlos(metrics, slos) {
|
|
4897
5163
|
const results = slos.map((slo) => check(slo, metrics[slo.metric]));
|
|
@@ -4978,70 +5244,445 @@ function scoreContinuity(pair, checks, options = {}) {
|
|
|
4978
5244
|
if (checks.length === 0) {
|
|
4979
5245
|
throw new Error("scoreContinuity: at least 1 check required");
|
|
4980
5246
|
}
|
|
4981
|
-
const passThreshold = options.passThreshold ?? 0.8;
|
|
4982
|
-
const results = checks.map((c) => {
|
|
4983
|
-
const raw = c.score(pair);
|
|
4984
|
-
const clamped = Number.isFinite(raw) ? Math.max(0, Math.min(1, raw)) : 0;
|
|
4985
|
-
return { id: c.id, description: c.description, score: clamped, pass: clamped >= passThreshold };
|
|
4986
|
-
});
|
|
4987
|
-
const overallScore = results.reduce((a, r) => a + r.score, 0) / results.length;
|
|
4988
|
-
return { results, overallScore, pass: results.every((r) => r.pass) };
|
|
5247
|
+
const passThreshold = options.passThreshold ?? 0.8;
|
|
5248
|
+
const results = checks.map((c) => {
|
|
5249
|
+
const raw = c.score(pair);
|
|
5250
|
+
const clamped = Number.isFinite(raw) ? Math.max(0, Math.min(1, raw)) : 0;
|
|
5251
|
+
return { id: c.id, description: c.description, score: clamped, pass: clamped >= passThreshold };
|
|
5252
|
+
});
|
|
5253
|
+
const overallScore = results.reduce((a, r) => a + r.score, 0) / results.length;
|
|
5254
|
+
return { results, overallScore, pass: results.every((r) => r.pass) };
|
|
5255
|
+
}
|
|
5256
|
+
function keyPreserved(key) {
|
|
5257
|
+
return {
|
|
5258
|
+
id: `preserved(${key})`,
|
|
5259
|
+
description: `"${key}" unchanged from before to after`,
|
|
5260
|
+
score: ({ before, after }) => before[key] !== void 0 && before[key] === after[key] ? 1 : 0
|
|
5261
|
+
};
|
|
5262
|
+
}
|
|
5263
|
+
function collectionPreserved(key, minRatio = 1) {
|
|
5264
|
+
return {
|
|
5265
|
+
id: `collection-preserved(${key})`,
|
|
5266
|
+
description: `"${key}" length \u2265 ${minRatio} \xD7 prior length`,
|
|
5267
|
+
score: ({ before, after }) => {
|
|
5268
|
+
const b = before[key];
|
|
5269
|
+
const a = after[key];
|
|
5270
|
+
if (!Array.isArray(b) || !Array.isArray(a)) return 0;
|
|
5271
|
+
if (b.length === 0) return a.length === 0 ? 1 : 1;
|
|
5272
|
+
return Math.min(1, a.length / (b.length * minRatio));
|
|
5273
|
+
}
|
|
5274
|
+
};
|
|
5275
|
+
}
|
|
5276
|
+
function statusAdvanced(key, progression) {
|
|
5277
|
+
return {
|
|
5278
|
+
id: `status-advanced(${key})`,
|
|
5279
|
+
description: `"${key}" progressed along ${progression.join("\u2192")}`,
|
|
5280
|
+
score: ({ before, after }) => {
|
|
5281
|
+
const bi = progression.indexOf(String(before[key]));
|
|
5282
|
+
const ai2 = progression.indexOf(String(after[key]));
|
|
5283
|
+
if (bi === -1 || ai2 === -1) return 0;
|
|
5284
|
+
return ai2 >= bi ? 1 : 0;
|
|
5285
|
+
}
|
|
5286
|
+
};
|
|
5287
|
+
}
|
|
5288
|
+
|
|
5289
|
+
// src/ui-finding.ts
|
|
5290
|
+
var UI_LENSES = [
|
|
5291
|
+
"consistency",
|
|
5292
|
+
"hierarchy",
|
|
5293
|
+
"layout",
|
|
5294
|
+
"ux-flow",
|
|
5295
|
+
"duplication",
|
|
5296
|
+
"accessibility",
|
|
5297
|
+
"responsive",
|
|
5298
|
+
"states",
|
|
5299
|
+
"content",
|
|
5300
|
+
"interaction",
|
|
5301
|
+
"performance-perceived",
|
|
5302
|
+
"other"
|
|
5303
|
+
];
|
|
5304
|
+
var UI_FINDING_SEVERITIES = [
|
|
5305
|
+
"critical",
|
|
5306
|
+
"high",
|
|
5307
|
+
"med",
|
|
5308
|
+
"low"
|
|
5309
|
+
];
|
|
5310
|
+
|
|
5311
|
+
// src/trace-contracts.ts
|
|
5312
|
+
function isSerializedRegex(v) {
|
|
5313
|
+
return typeof v === "object" && v !== null && typeof v.$regex === "string" && typeof v.flags === "string";
|
|
5314
|
+
}
|
|
5315
|
+
function matchText(actual, matcher) {
|
|
5316
|
+
if (typeof actual !== "string") return false;
|
|
5317
|
+
if (typeof matcher === "string") return actual === matcher;
|
|
5318
|
+
if (matcher instanceof RegExp) return matcher.test(actual);
|
|
5319
|
+
return new RegExp(matcher.$regex, matcher.flags).test(actual);
|
|
5320
|
+
}
|
|
5321
|
+
function resolveToolName(span) {
|
|
5322
|
+
if (typeof span.toolName === "string") return span.toolName;
|
|
5323
|
+
const fromAttr = span.attributes?.["tool.name"] ?? span.attributes?.["toolName"];
|
|
5324
|
+
if (typeof fromAttr === "string") return fromAttr;
|
|
5325
|
+
if (span.kind === "tool" && typeof span.name === "string") return span.name;
|
|
5326
|
+
return void 0;
|
|
5327
|
+
}
|
|
5328
|
+
function assertPredicate(value, where) {
|
|
5329
|
+
if (value === null || typeof value !== "object") {
|
|
5330
|
+
throw new ValidationError(`${where}: predicate must be an object, got ${typeof value}`);
|
|
5331
|
+
}
|
|
5332
|
+
const p = value;
|
|
5333
|
+
for (const field of ["name", "tool"]) {
|
|
5334
|
+
const m = p[field];
|
|
5335
|
+
if (m !== void 0 && typeof m !== "string" && !(m instanceof RegExp) && !isSerializedRegex(m)) {
|
|
5336
|
+
throw new ValidationError(`${where}: "${field}" must be string | RegExp | SerializedRegex`);
|
|
5337
|
+
}
|
|
5338
|
+
}
|
|
5339
|
+
if (p.attr !== void 0 && (p.attr === null || typeof p.attr !== "object" || Array.isArray(p.attr))) {
|
|
5340
|
+
throw new ValidationError(`${where}: "attr" must be a plain object`);
|
|
5341
|
+
}
|
|
5342
|
+
if (p.requiresCustom && typeof p.custom !== "function") {
|
|
5343
|
+
throw new ValidationError(
|
|
5344
|
+
`${where}: rule was built with a custom predicate function, which does not survive JSON serialization \u2014 re-attach \`custom\` after deserializing or drop the rule`
|
|
5345
|
+
);
|
|
5346
|
+
}
|
|
5347
|
+
const hasAttr = p.attr !== void 0 && Object.keys(p.attr).length > 0;
|
|
5348
|
+
if (p.name === void 0 && p.tool === void 0 && !hasAttr && typeof p.custom !== "function") {
|
|
5349
|
+
throw new ValidationError(
|
|
5350
|
+
`${where}: empty predicate would match every span \u2014 specify name, tool, attr, or custom`
|
|
5351
|
+
);
|
|
5352
|
+
}
|
|
5353
|
+
}
|
|
5354
|
+
function matchSpan(span, predicate) {
|
|
5355
|
+
assertPredicate(predicate, "matchSpan");
|
|
5356
|
+
if (predicate.name !== void 0 && !matchText(span.name, predicate.name)) return false;
|
|
5357
|
+
if (predicate.tool !== void 0 && !matchText(resolveToolName(span), predicate.tool))
|
|
5358
|
+
return false;
|
|
5359
|
+
if (predicate.attr !== void 0) {
|
|
5360
|
+
for (const [key, expected] of Object.entries(predicate.attr)) {
|
|
5361
|
+
const actual = span.attributes?.[key];
|
|
5362
|
+
if (expected instanceof RegExp || isSerializedRegex(expected)) {
|
|
5363
|
+
if (!matchText(typeof actual === "string" ? actual : void 0, expected)) {
|
|
5364
|
+
return false;
|
|
5365
|
+
}
|
|
5366
|
+
} else if (actual !== expected) {
|
|
5367
|
+
return false;
|
|
5368
|
+
}
|
|
5369
|
+
}
|
|
5370
|
+
}
|
|
5371
|
+
if (predicate.custom !== void 0 && !predicate.custom(span)) return false;
|
|
5372
|
+
return true;
|
|
5373
|
+
}
|
|
5374
|
+
function describeMatcher(m) {
|
|
5375
|
+
if (typeof m === "string") return m;
|
|
5376
|
+
if (m instanceof RegExp) return `/${m.source}/${m.flags}`;
|
|
5377
|
+
return `/${m.$regex}/${m.flags}`;
|
|
5378
|
+
}
|
|
5379
|
+
function describePredicate(p) {
|
|
5380
|
+
const parts = [];
|
|
5381
|
+
if (p.name !== void 0) parts.push(`name=${describeMatcher(p.name)}`);
|
|
5382
|
+
if (p.tool !== void 0) parts.push(`tool=${describeMatcher(p.tool)}`);
|
|
5383
|
+
if (p.attr !== void 0) {
|
|
5384
|
+
for (const [k, v] of Object.entries(p.attr)) {
|
|
5385
|
+
parts.push(
|
|
5386
|
+
`attr.${k}=${v instanceof RegExp || isSerializedRegex(v) ? describeMatcher(v) : JSON.stringify(v)}`
|
|
5387
|
+
);
|
|
5388
|
+
}
|
|
5389
|
+
}
|
|
5390
|
+
if (typeof p.custom === "function") parts.push(`custom=${p.custom.name || "fn"}`);
|
|
5391
|
+
return parts.join(",");
|
|
5392
|
+
}
|
|
5393
|
+
function normalizeMatcher(m) {
|
|
5394
|
+
return m instanceof RegExp ? { $regex: m.source, flags: m.flags } : m;
|
|
5395
|
+
}
|
|
5396
|
+
function normalizePredicate(p, where) {
|
|
5397
|
+
assertPredicate(p, where);
|
|
5398
|
+
const out = {};
|
|
5399
|
+
if (p.name !== void 0) out.name = normalizeMatcher(p.name);
|
|
5400
|
+
if (p.tool !== void 0) out.tool = normalizeMatcher(p.tool);
|
|
5401
|
+
if (p.attr !== void 0) {
|
|
5402
|
+
out.attr = Object.fromEntries(
|
|
5403
|
+
Object.entries(p.attr).map(([k, v]) => [k, v instanceof RegExp ? normalizeMatcher(v) : v])
|
|
5404
|
+
);
|
|
5405
|
+
}
|
|
5406
|
+
if (typeof p.custom === "function") {
|
|
5407
|
+
out.custom = p.custom;
|
|
5408
|
+
out.requiresCustom = true;
|
|
5409
|
+
}
|
|
5410
|
+
return out;
|
|
5411
|
+
}
|
|
5412
|
+
var TraceContractBuilder = class {
|
|
5413
|
+
constructor(name) {
|
|
5414
|
+
this.name = name;
|
|
5415
|
+
}
|
|
5416
|
+
name;
|
|
5417
|
+
rules = [];
|
|
5418
|
+
/** Every span in the trace must satisfy `p`. */
|
|
5419
|
+
always(p, label) {
|
|
5420
|
+
const np = normalizePredicate(p, `traceContract("${this.name}").always`);
|
|
5421
|
+
return this.add({ kind: "always", label: label ?? `always(${describePredicate(np)})`, p: np });
|
|
5422
|
+
}
|
|
5423
|
+
/** No span in the trace may satisfy `p`. */
|
|
5424
|
+
never(p, label) {
|
|
5425
|
+
const np = normalizePredicate(p, `traceContract("${this.name}").never`);
|
|
5426
|
+
return this.add({ kind: "never", label: label ?? `never(${describePredicate(np)})`, p: np });
|
|
5427
|
+
}
|
|
5428
|
+
/** At least one span in the trace must satisfy `p`. */
|
|
5429
|
+
eventually(p, label) {
|
|
5430
|
+
const np = normalizePredicate(p, `traceContract("${this.name}").eventually`);
|
|
5431
|
+
return this.add({
|
|
5432
|
+
kind: "eventually",
|
|
5433
|
+
label: label ?? `eventually(${describePredicate(np)})`,
|
|
5434
|
+
p: np
|
|
5435
|
+
});
|
|
5436
|
+
}
|
|
5437
|
+
/** Every `b`-match must have a strictly earlier `a`-match. */
|
|
5438
|
+
precedes(a, b, label) {
|
|
5439
|
+
const na = normalizePredicate(a, `traceContract("${this.name}").precedes (a)`);
|
|
5440
|
+
const nb = normalizePredicate(b, `traceContract("${this.name}").precedes (b)`);
|
|
5441
|
+
return this.add({
|
|
5442
|
+
kind: "precedes",
|
|
5443
|
+
label: label ?? `precedes(${describePredicate(na)} -> ${describePredicate(nb)})`,
|
|
5444
|
+
a: na,
|
|
5445
|
+
b: nb
|
|
5446
|
+
});
|
|
5447
|
+
}
|
|
5448
|
+
/** Every `p`-match is a violation unless a strictly earlier `prior`-match exists. */
|
|
5449
|
+
neverUnless(p, prior, label) {
|
|
5450
|
+
const np = normalizePredicate(p, `traceContract("${this.name}").neverUnless (p)`);
|
|
5451
|
+
const nprior = normalizePredicate(prior, `traceContract("${this.name}").neverUnless (prior)`);
|
|
5452
|
+
return this.add({
|
|
5453
|
+
kind: "neverUnless",
|
|
5454
|
+
label: label ?? `neverUnless(${describePredicate(np)} unless ${describePredicate(nprior)})`,
|
|
5455
|
+
p: np,
|
|
5456
|
+
prior: nprior
|
|
5457
|
+
});
|
|
5458
|
+
}
|
|
5459
|
+
build() {
|
|
5460
|
+
if (this.rules.length === 0) {
|
|
5461
|
+
throw new ValidationError(
|
|
5462
|
+
`traceContract("${this.name}").build(): no rules \u2014 an empty contract would vacuously pass`
|
|
5463
|
+
);
|
|
5464
|
+
}
|
|
5465
|
+
return { name: this.name, rules: [...this.rules] };
|
|
5466
|
+
}
|
|
5467
|
+
add(rule) {
|
|
5468
|
+
let label = rule.label;
|
|
5469
|
+
let n = 2;
|
|
5470
|
+
while (this.rules.some((r) => r.label === label)) {
|
|
5471
|
+
label = `${rule.label} #${n}`;
|
|
5472
|
+
n += 1;
|
|
5473
|
+
}
|
|
5474
|
+
this.rules.push({ ...rule, label });
|
|
5475
|
+
return this;
|
|
5476
|
+
}
|
|
5477
|
+
};
|
|
5478
|
+
function traceContract(name) {
|
|
5479
|
+
if (typeof name !== "string" || name.length === 0) {
|
|
5480
|
+
throw new ValidationError("traceContract: name must be a non-empty string");
|
|
5481
|
+
}
|
|
5482
|
+
return new TraceContractBuilder(name);
|
|
5483
|
+
}
|
|
5484
|
+
function assertContract(contract) {
|
|
5485
|
+
if (typeof contract?.name !== "string" || contract.name.length === 0) {
|
|
5486
|
+
throw new ValidationError("evaluateTraceContract: contract.name must be a non-empty string");
|
|
5487
|
+
}
|
|
5488
|
+
if (!Array.isArray(contract.rules) || contract.rules.length === 0) {
|
|
5489
|
+
throw new ValidationError(
|
|
5490
|
+
`evaluateTraceContract: contract "${contract.name}" has no rules \u2014 an empty contract would vacuously pass`
|
|
5491
|
+
);
|
|
5492
|
+
}
|
|
5493
|
+
const seen = /* @__PURE__ */ new Set();
|
|
5494
|
+
for (const rule of contract.rules) {
|
|
5495
|
+
const where = `contract "${contract.name}" rule "${rule?.label ?? "<unlabeled>"}"`;
|
|
5496
|
+
if (typeof rule?.label !== "string" || rule.label.length === 0) {
|
|
5497
|
+
throw new ValidationError(`${where}: label must be a non-empty string`);
|
|
5498
|
+
}
|
|
5499
|
+
if (seen.has(rule.label)) {
|
|
5500
|
+
throw new ValidationError(`${where}: duplicate label would collapse per-rule scores`);
|
|
5501
|
+
}
|
|
5502
|
+
seen.add(rule.label);
|
|
5503
|
+
switch (rule.kind) {
|
|
5504
|
+
case "always":
|
|
5505
|
+
case "never":
|
|
5506
|
+
case "eventually":
|
|
5507
|
+
assertPredicate(rule.p, `${where} (p)`);
|
|
5508
|
+
break;
|
|
5509
|
+
case "precedes":
|
|
5510
|
+
assertPredicate(rule.a, `${where} (a)`);
|
|
5511
|
+
assertPredicate(rule.b, `${where} (b)`);
|
|
5512
|
+
break;
|
|
5513
|
+
case "neverUnless":
|
|
5514
|
+
assertPredicate(rule.p, `${where} (p)`);
|
|
5515
|
+
assertPredicate(rule.prior, `${where} (prior)`);
|
|
5516
|
+
break;
|
|
5517
|
+
default:
|
|
5518
|
+
throw new ValidationError(
|
|
5519
|
+
`${where}: unknown rule kind "${String(rule.kind)}"`
|
|
5520
|
+
);
|
|
5521
|
+
}
|
|
5522
|
+
}
|
|
5523
|
+
}
|
|
5524
|
+
function orderSpans(spans) {
|
|
5525
|
+
const timed = spans.filter(
|
|
5526
|
+
(s) => typeof s.startedAt === "number" && Number.isFinite(s.startedAt)
|
|
5527
|
+
).length;
|
|
5528
|
+
if (timed === spans.length) {
|
|
5529
|
+
return spans.map((span, i) => ({ span, i })).sort((x, y) => x.span.startedAt - y.span.startedAt || x.i - y.i).map((x) => x.span);
|
|
5530
|
+
}
|
|
5531
|
+
if (timed === 0) return [...spans];
|
|
5532
|
+
throw new ValidationError(
|
|
5533
|
+
`evaluateTraceContract: ${timed}/${spans.length} spans carry a finite startedAt \u2014 mixed timestamps make ordering ambiguous; stamp every span or none`
|
|
5534
|
+
);
|
|
5535
|
+
}
|
|
5536
|
+
function spanRef(span, index) {
|
|
5537
|
+
return span.spanId ?? `#${index}`;
|
|
5538
|
+
}
|
|
5539
|
+
function checkGuarded(args) {
|
|
5540
|
+
const out = [];
|
|
5541
|
+
let guardSeen = false;
|
|
5542
|
+
for (let i = 0; i < args.ordered.length; i++) {
|
|
5543
|
+
const span = args.ordered[i];
|
|
5544
|
+
if (!guardSeen && matchSpan(span, args.guarded)) {
|
|
5545
|
+
out.push({ rule: args.label, spanId: span.spanId, detail: args.detail(spanRef(span, i)) });
|
|
5546
|
+
}
|
|
5547
|
+
if (matchSpan(span, args.guard)) guardSeen = true;
|
|
5548
|
+
}
|
|
5549
|
+
return out;
|
|
5550
|
+
}
|
|
5551
|
+
function checkRule(rule, ordered) {
|
|
5552
|
+
switch (rule.kind) {
|
|
5553
|
+
case "always": {
|
|
5554
|
+
const out = [];
|
|
5555
|
+
for (let i = 0; i < ordered.length; i++) {
|
|
5556
|
+
const span = ordered[i];
|
|
5557
|
+
if (!matchSpan(span, rule.p)) {
|
|
5558
|
+
out.push({
|
|
5559
|
+
rule: rule.label,
|
|
5560
|
+
spanId: span.spanId,
|
|
5561
|
+
detail: `span ${spanRef(span, i)} ("${span.name ?? ""}") fails always(${describePredicate(rule.p)})`
|
|
5562
|
+
});
|
|
5563
|
+
}
|
|
5564
|
+
}
|
|
5565
|
+
return out;
|
|
5566
|
+
}
|
|
5567
|
+
case "never": {
|
|
5568
|
+
const out = [];
|
|
5569
|
+
for (let i = 0; i < ordered.length; i++) {
|
|
5570
|
+
const span = ordered[i];
|
|
5571
|
+
if (matchSpan(span, rule.p)) {
|
|
5572
|
+
out.push({
|
|
5573
|
+
rule: rule.label,
|
|
5574
|
+
spanId: span.spanId,
|
|
5575
|
+
detail: `span ${spanRef(span, i)} ("${span.name ?? ""}") matches never(${describePredicate(rule.p)})`
|
|
5576
|
+
});
|
|
5577
|
+
}
|
|
5578
|
+
}
|
|
5579
|
+
return out;
|
|
5580
|
+
}
|
|
5581
|
+
case "eventually": {
|
|
5582
|
+
const hit = ordered.some((span) => matchSpan(span, rule.p));
|
|
5583
|
+
return hit ? [] : [
|
|
5584
|
+
{
|
|
5585
|
+
rule: rule.label,
|
|
5586
|
+
detail: `no span matches eventually(${describePredicate(rule.p)}) over ${ordered.length} span(s)`
|
|
5587
|
+
}
|
|
5588
|
+
];
|
|
5589
|
+
}
|
|
5590
|
+
case "precedes":
|
|
5591
|
+
return checkGuarded({
|
|
5592
|
+
label: rule.label,
|
|
5593
|
+
ordered,
|
|
5594
|
+
guard: rule.a,
|
|
5595
|
+
guarded: rule.b,
|
|
5596
|
+
detail: (ref) => `span ${ref} matches ${describePredicate(rule.b)} with no earlier ${describePredicate(rule.a)} match`
|
|
5597
|
+
});
|
|
5598
|
+
case "neverUnless":
|
|
5599
|
+
return checkGuarded({
|
|
5600
|
+
label: rule.label,
|
|
5601
|
+
ordered,
|
|
5602
|
+
guard: rule.prior,
|
|
5603
|
+
guarded: rule.p,
|
|
5604
|
+
detail: (ref) => `span ${ref} matches ${describePredicate(rule.p)} with no earlier ${describePredicate(rule.prior)} match`
|
|
5605
|
+
});
|
|
5606
|
+
}
|
|
4989
5607
|
}
|
|
4990
|
-
function
|
|
5608
|
+
function evaluateTraceContract(contract, spans) {
|
|
5609
|
+
assertContract(contract);
|
|
5610
|
+
const ordered = orderSpans(spans);
|
|
5611
|
+
const scores2 = {};
|
|
5612
|
+
const violations = [];
|
|
5613
|
+
for (const rule of contract.rules) {
|
|
5614
|
+
const ruleViolations = checkRule(rule, ordered);
|
|
5615
|
+
scores2[rule.label] = ruleViolations.length === 0 ? 1 : 0;
|
|
5616
|
+
violations.push(...ruleViolations);
|
|
5617
|
+
}
|
|
5618
|
+
const ruleCount = contract.rules.length;
|
|
5619
|
+
const passCount = Object.values(scores2).filter((s) => s === 1).length;
|
|
4991
5620
|
return {
|
|
4992
|
-
|
|
4993
|
-
|
|
4994
|
-
score:
|
|
5621
|
+
contract: contract.name,
|
|
5622
|
+
valid: passCount === ruleCount,
|
|
5623
|
+
score: passCount / ruleCount,
|
|
5624
|
+
scores: scores2,
|
|
5625
|
+
violations,
|
|
5626
|
+
notes: `${passCount}/${ruleCount} rules passed`
|
|
4995
5627
|
};
|
|
4996
5628
|
}
|
|
4997
|
-
function
|
|
4998
|
-
|
|
4999
|
-
|
|
5000
|
-
|
|
5001
|
-
|
|
5002
|
-
|
|
5003
|
-
|
|
5004
|
-
|
|
5005
|
-
if (b.length === 0) return a.length === 0 ? 1 : 1;
|
|
5006
|
-
return Math.min(1, a.length / (b.length * minRatio));
|
|
5007
|
-
}
|
|
5008
|
-
};
|
|
5629
|
+
function checkTraceContracts(spans, contracts) {
|
|
5630
|
+
if (contracts.length === 0) {
|
|
5631
|
+
throw new ValidationError(
|
|
5632
|
+
"checkTraceContracts: empty contract list would vacuously pass \u2014 supply at least one contract"
|
|
5633
|
+
);
|
|
5634
|
+
}
|
|
5635
|
+
const verdicts = contracts.map((c) => evaluateTraceContract(c, spans));
|
|
5636
|
+
return { verdicts, allValid: verdicts.every((v) => v.valid) };
|
|
5009
5637
|
}
|
|
5010
|
-
function
|
|
5638
|
+
function contractJudge(contracts, opts) {
|
|
5639
|
+
if (contracts.length === 0) {
|
|
5640
|
+
throw new ValidationError("contractJudge: at least one contract required");
|
|
5641
|
+
}
|
|
5642
|
+
for (const c of contracts) assertContract(c);
|
|
5643
|
+
const names = /* @__PURE__ */ new Set();
|
|
5644
|
+
for (const c of contracts) {
|
|
5645
|
+
if (names.has(c.name)) {
|
|
5646
|
+
throw new ValidationError(
|
|
5647
|
+
`contractJudge: duplicate contract name "${c.name}" would collapse judge dimensions`
|
|
5648
|
+
);
|
|
5649
|
+
}
|
|
5650
|
+
names.add(c.name);
|
|
5651
|
+
}
|
|
5652
|
+
if (typeof opts?.spans !== "function") {
|
|
5653
|
+
throw new ValidationError("contractJudge: opts.spans extraction function is required");
|
|
5654
|
+
}
|
|
5655
|
+
const dimensions = contracts.map((c) => ({
|
|
5656
|
+
key: c.name,
|
|
5657
|
+
description: c.rules.map((r) => r.label).join("; ")
|
|
5658
|
+
}));
|
|
5011
5659
|
return {
|
|
5012
|
-
|
|
5013
|
-
|
|
5014
|
-
score
|
|
5015
|
-
const
|
|
5016
|
-
|
|
5017
|
-
|
|
5018
|
-
|
|
5660
|
+
name: opts.name ?? "trace-contracts",
|
|
5661
|
+
dimensions,
|
|
5662
|
+
score({ artifact, scenario }) {
|
|
5663
|
+
const spans = opts.spans({ artifact, scenario });
|
|
5664
|
+
if (!Array.isArray(spans)) {
|
|
5665
|
+
throw new ValidationError(
|
|
5666
|
+
`contractJudge: spans() must return a span array, got ${typeof spans}`
|
|
5667
|
+
);
|
|
5668
|
+
}
|
|
5669
|
+
const { verdicts } = checkTraceContracts(spans, contracts);
|
|
5670
|
+
const dims = {};
|
|
5671
|
+
const violations = [];
|
|
5672
|
+
for (const v of verdicts) {
|
|
5673
|
+
dims[v.contract] = v.score;
|
|
5674
|
+
violations.push(...v.violations);
|
|
5675
|
+
}
|
|
5676
|
+
const composite = verdicts.reduce((acc, v) => acc + v.score, 0) / verdicts.length;
|
|
5677
|
+
return {
|
|
5678
|
+
dimensions: dims,
|
|
5679
|
+
composite,
|
|
5680
|
+
notes: violations.length === 0 ? "all trace contracts satisfied" : violations.slice(0, 8).map((x) => `${x.rule}: ${x.detail}`).join("\n")
|
|
5681
|
+
};
|
|
5019
5682
|
}
|
|
5020
5683
|
};
|
|
5021
5684
|
}
|
|
5022
5685
|
|
|
5023
|
-
// src/ui-finding.ts
|
|
5024
|
-
var UI_LENSES = [
|
|
5025
|
-
"consistency",
|
|
5026
|
-
"hierarchy",
|
|
5027
|
-
"layout",
|
|
5028
|
-
"ux-flow",
|
|
5029
|
-
"duplication",
|
|
5030
|
-
"accessibility",
|
|
5031
|
-
"responsive",
|
|
5032
|
-
"states",
|
|
5033
|
-
"content",
|
|
5034
|
-
"interaction",
|
|
5035
|
-
"performance-perceived",
|
|
5036
|
-
"other"
|
|
5037
|
-
];
|
|
5038
|
-
var UI_FINDING_SEVERITIES = [
|
|
5039
|
-
"critical",
|
|
5040
|
-
"high",
|
|
5041
|
-
"med",
|
|
5042
|
-
"low"
|
|
5043
|
-
];
|
|
5044
|
-
|
|
5045
5686
|
// src/behavior-dsl.ts
|
|
5046
5687
|
var BehaviorAssertion = class {
|
|
5047
5688
|
constructor(store, runId) {
|
|
@@ -5109,6 +5750,25 @@ var BehaviorAssertion = class {
|
|
|
5109
5750
|
}
|
|
5110
5751
|
};
|
|
5111
5752
|
}
|
|
5753
|
+
/** Evaluate a finite-trace temporal contract (`traceContract(...)`) over
|
|
5754
|
+
* this run's span sequence. See `trace-contracts.ts` for the operators. */
|
|
5755
|
+
toSatisfyContract(contract) {
|
|
5756
|
+
return {
|
|
5757
|
+
label: `agent(${this.runId}).toSatisfyContract(${contract.name})`,
|
|
5758
|
+
check: async () => {
|
|
5759
|
+
const spans = await this.store.spans({ runId: this.runId });
|
|
5760
|
+
const verdict = evaluateTraceContract(contract, spans);
|
|
5761
|
+
if (verdict.valid) {
|
|
5762
|
+
return { ok: true, detail: `contract "${contract.name}": ${verdict.notes}` };
|
|
5763
|
+
}
|
|
5764
|
+
return {
|
|
5765
|
+
ok: false,
|
|
5766
|
+
detail: verdict.violations.map((v) => `${v.rule}: ${v.detail}`).join("; "),
|
|
5767
|
+
evidence: verdict.violations.find((v) => v.spanId)?.spanId
|
|
5768
|
+
};
|
|
5769
|
+
}
|
|
5770
|
+
};
|
|
5771
|
+
}
|
|
5112
5772
|
toNeverCall(toolName) {
|
|
5113
5773
|
return {
|
|
5114
5774
|
label: `agent(${this.runId}).toNeverCall(${toolName})`,
|
|
@@ -5711,90 +6371,6 @@ async function promptBisect(options) {
|
|
|
5711
6371
|
};
|
|
5712
6372
|
}
|
|
5713
6373
|
|
|
5714
|
-
// src/counterfactual.ts
|
|
5715
|
-
async function runCounterfactual(store, originalRunId, mutation, runner) {
|
|
5716
|
-
const originalRun = await store.getRun(originalRunId);
|
|
5717
|
-
if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
|
|
5718
|
-
const trajectory = await buildTrajectory(store, originalRunId);
|
|
5719
|
-
if (mutation.at < 0 || mutation.at >= trajectory.steps.length) {
|
|
5720
|
-
throw new ValidationError(
|
|
5721
|
-
`counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`
|
|
5722
|
-
);
|
|
5723
|
-
}
|
|
5724
|
-
const targetStep = trajectory.steps[mutation.at];
|
|
5725
|
-
const mutatedStep = applyMutation(targetStep, mutation);
|
|
5726
|
-
const cfEmitter = new TraceEmitter(store);
|
|
5727
|
-
await cfEmitter.startRun({
|
|
5728
|
-
scenarioId: originalRun.scenarioId,
|
|
5729
|
-
variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
|
|
5730
|
-
projectId: originalRun.projectId,
|
|
5731
|
-
parentRunId: originalRunId,
|
|
5732
|
-
layer: "meta",
|
|
5733
|
-
tags: { counterfactual: "true", mutationKind: mutation.kind, mutationAt: String(mutation.at) }
|
|
5734
|
-
});
|
|
5735
|
-
await runner.executeFrom(
|
|
5736
|
-
{
|
|
5737
|
-
originalRunId,
|
|
5738
|
-
originalTrajectory: trajectory,
|
|
5739
|
-
prefix: trajectory.steps.slice(0, mutation.at),
|
|
5740
|
-
mutation,
|
|
5741
|
-
mutatedStep
|
|
5742
|
-
},
|
|
5743
|
-
cfEmitter
|
|
5744
|
-
);
|
|
5745
|
-
const counterfactual = await store.getRun(cfEmitter.runId);
|
|
5746
|
-
const delta = {
|
|
5747
|
-
originalOutcomeScore: originalRun.outcome?.score ?? null,
|
|
5748
|
-
counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
|
|
5749
|
-
deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
|
|
5750
|
-
};
|
|
5751
|
-
return { counterfactualRunId: cfEmitter.runId, originalRunId, mutation, delta };
|
|
5752
|
-
}
|
|
5753
|
-
function applyMutation(step, mutation) {
|
|
5754
|
-
if (mutation.kind === "swap-model" && step.span.kind === "llm") {
|
|
5755
|
-
const llm = step.span;
|
|
5756
|
-
return { ...step, span: { ...llm, model: mutation.newModel } };
|
|
5757
|
-
}
|
|
5758
|
-
if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
|
|
5759
|
-
const tool = step.span;
|
|
5760
|
-
return { ...step, span: { ...tool, result: mutation.newResult } };
|
|
5761
|
-
}
|
|
5762
|
-
if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
|
|
5763
|
-
const llm = step.span;
|
|
5764
|
-
return {
|
|
5765
|
-
...step,
|
|
5766
|
-
span: {
|
|
5767
|
-
...llm,
|
|
5768
|
-
messages: [{ role: "system", content: mutation.content }, ...llm.messages]
|
|
5769
|
-
}
|
|
5770
|
-
};
|
|
5771
|
-
}
|
|
5772
|
-
if (mutation.kind === "custom") return mutation.apply(step);
|
|
5773
|
-
return step;
|
|
5774
|
-
}
|
|
5775
|
-
function attributeCounterfactuals(results) {
|
|
5776
|
-
const grouped = /* @__PURE__ */ new Map();
|
|
5777
|
-
for (const r of results) {
|
|
5778
|
-
const arr = grouped.get(r.mutation.kind) ?? [];
|
|
5779
|
-
arr.push(r);
|
|
5780
|
-
grouped.set(r.mutation.kind, arr);
|
|
5781
|
-
}
|
|
5782
|
-
const out = [];
|
|
5783
|
-
for (const [kind, items] of grouped) {
|
|
5784
|
-
const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
|
|
5785
|
-
if (deltas.length === 0) continue;
|
|
5786
|
-
const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
|
|
5787
|
-
const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
|
|
5788
|
-
out.push({
|
|
5789
|
-
mutationKind: kind,
|
|
5790
|
-
n: deltas.length,
|
|
5791
|
-
meanAbsDelta: meanAbs,
|
|
5792
|
-
meanSignedDelta: meanSigned
|
|
5793
|
-
});
|
|
5794
|
-
}
|
|
5795
|
-
return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
|
|
5796
|
-
}
|
|
5797
|
-
|
|
5798
6374
|
// src/cross-trace-diff.ts
|
|
5799
6375
|
async function crossTraceDiff(store, runA, runB, options = {}) {
|
|
5800
6376
|
const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
|
|
@@ -6986,14 +7562,18 @@ function aggregate(layers, results, startedAt, startedAtMs) {
|
|
|
6986
7562
|
}
|
|
6987
7563
|
}
|
|
6988
7564
|
const finishedAtMs = Date.now();
|
|
7565
|
+
const allPass = ranAnyScoredLayer && !anyScoredLayerFailed && failCount === 0 && errorCount === 0;
|
|
7566
|
+
const blendedScore = scoredWeightSum > 0 ? scoredWeightedTotal / scoredWeightSum : 0;
|
|
6989
7567
|
return {
|
|
6990
7568
|
layers: results,
|
|
6991
7569
|
passCount,
|
|
6992
7570
|
failCount,
|
|
6993
7571
|
skippedCount,
|
|
6994
7572
|
errorCount,
|
|
6995
|
-
allPass
|
|
6996
|
-
blendedScore
|
|
7573
|
+
allPass,
|
|
7574
|
+
blendedScore,
|
|
7575
|
+
valid: allPass,
|
|
7576
|
+
score: blendedScore,
|
|
6997
7577
|
durationMs: finishedAtMs - startedAtMs,
|
|
6998
7578
|
startedAt,
|
|
6999
7579
|
finishedAt: new Date(finishedAtMs).toISOString()
|
|
@@ -8208,69 +8788,6 @@ function fmt(x) {
|
|
|
8208
8788
|
return x.toFixed(4);
|
|
8209
8789
|
}
|
|
8210
8790
|
|
|
8211
|
-
// src/judge-retry.ts
|
|
8212
|
-
var DEFAULT_MAX_ATTEMPTS = 3;
|
|
8213
|
-
var DEFAULT_TIMEOUT_MS = 3e5;
|
|
8214
|
-
function sleep(ms) {
|
|
8215
|
-
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
8216
|
-
}
|
|
8217
|
-
async function withJudgeRetry(judgeFn, policy = {}) {
|
|
8218
|
-
const maxAttempts = policy.maxAttempts ?? DEFAULT_MAX_ATTEMPTS;
|
|
8219
|
-
const timeoutMs = policy.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
8220
|
-
const backoff = policy.backoffMs ?? backoffMs;
|
|
8221
|
-
const isRetryable = policy.isRetryable ?? isTransientLlmError;
|
|
8222
|
-
const models = policy.models && policy.models.length > 0 ? policy.models : [void 0];
|
|
8223
|
-
let totalAttempts = 0;
|
|
8224
|
-
const attemptErrors = [];
|
|
8225
|
-
let lastError;
|
|
8226
|
-
for (const model of models) {
|
|
8227
|
-
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
8228
|
-
totalAttempts += 1;
|
|
8229
|
-
const controller = new AbortController();
|
|
8230
|
-
const timer = setTimeout(() => controller.abort(new Error("TimeoutError")), timeoutMs);
|
|
8231
|
-
try {
|
|
8232
|
-
const value = await judgeFn(model, controller.signal);
|
|
8233
|
-
clearTimeout(timer);
|
|
8234
|
-
return {
|
|
8235
|
-
value,
|
|
8236
|
-
succeeded: true,
|
|
8237
|
-
attempts: totalAttempts,
|
|
8238
|
-
modelUsed: model,
|
|
8239
|
-
attemptErrors
|
|
8240
|
-
};
|
|
8241
|
-
} catch (err) {
|
|
8242
|
-
clearTimeout(timer);
|
|
8243
|
-
const errObj = err instanceof Error ? err : new Error(String(err));
|
|
8244
|
-
lastError = errObj;
|
|
8245
|
-
attemptErrors.push({
|
|
8246
|
-
attempt: totalAttempts,
|
|
8247
|
-
model: model ?? "(default)",
|
|
8248
|
-
error: errObj.message
|
|
8249
|
-
});
|
|
8250
|
-
if (!isRetryable(errObj)) {
|
|
8251
|
-
return {
|
|
8252
|
-
value: null,
|
|
8253
|
-
succeeded: false,
|
|
8254
|
-
attempts: totalAttempts,
|
|
8255
|
-
error: errObj,
|
|
8256
|
-
attemptErrors
|
|
8257
|
-
};
|
|
8258
|
-
}
|
|
8259
|
-
if (attempt < maxAttempts - 1) {
|
|
8260
|
-
await sleep(backoff(attempt));
|
|
8261
|
-
}
|
|
8262
|
-
}
|
|
8263
|
-
}
|
|
8264
|
-
}
|
|
8265
|
-
return {
|
|
8266
|
-
value: null,
|
|
8267
|
-
succeeded: false,
|
|
8268
|
-
attempts: totalAttempts,
|
|
8269
|
-
error: lastError,
|
|
8270
|
-
attemptErrors
|
|
8271
|
-
};
|
|
8272
|
-
}
|
|
8273
|
-
|
|
8274
8791
|
// src/orthogonality.ts
|
|
8275
8792
|
function passOrthogonality(input) {
|
|
8276
8793
|
const passes = input.passes;
|
|
@@ -8649,6 +9166,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
|
8649
9166
|
});
|
|
8650
9167
|
try {
|
|
8651
9168
|
const allScores = [];
|
|
9169
|
+
let failedJudges = 0;
|
|
8652
9170
|
for (let i = 0; i < judges.length; i++) {
|
|
8653
9171
|
const judge = judges[i];
|
|
8654
9172
|
const name = judgeNames[i] ?? `judge_${i}`;
|
|
@@ -8656,8 +9174,13 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
|
8656
9174
|
emitter: opts.emitter,
|
|
8657
9175
|
parentSpanId: ensembleSpan.span.spanId
|
|
8658
9176
|
});
|
|
8659
|
-
|
|
8660
|
-
|
|
9177
|
+
try {
|
|
9178
|
+
const scores2 = await tracedFn(tc, input);
|
|
9179
|
+
allScores.push(...scores2);
|
|
9180
|
+
} catch (err) {
|
|
9181
|
+
if (!(err instanceof JudgeParseError)) throw err;
|
|
9182
|
+
failedJudges++;
|
|
9183
|
+
}
|
|
8661
9184
|
}
|
|
8662
9185
|
const composite = allScores.length > 0 ? allScores.reduce((sum3, s) => sum3 + s.score, 0) / allScores.length : 0;
|
|
8663
9186
|
await ensembleSpan.end({
|
|
@@ -8665,6 +9188,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
|
|
|
8665
9188
|
"judge.ensemble_size": judges.length,
|
|
8666
9189
|
"judge.composite_score": composite,
|
|
8667
9190
|
"judge.total_dimensions": allScores.length,
|
|
9191
|
+
"judge.failed_judges": failedJudges,
|
|
8668
9192
|
"eval.phase": "judge"
|
|
8669
9193
|
}
|
|
8670
9194
|
});
|
|
@@ -9094,8 +9618,256 @@ function applyDomainPatch(p, sectionId, newBody) {
|
|
|
9094
9618
|
function sectionHash(section) {
|
|
9095
9619
|
return surfaceContentHash(JSON.stringify({ title: section.title, body: section.body }));
|
|
9096
9620
|
}
|
|
9621
|
+
|
|
9622
|
+
// src/cost-report.ts
|
|
9623
|
+
function costReport(ledger) {
|
|
9624
|
+
const summary = ledger.summary();
|
|
9625
|
+
const perModel = /* @__PURE__ */ new Map();
|
|
9626
|
+
for (const entry of ledger.list()) {
|
|
9627
|
+
const roll = perModel.get(entry.model) ?? {
|
|
9628
|
+
model: entry.model,
|
|
9629
|
+
usd: 0,
|
|
9630
|
+
entries: 0,
|
|
9631
|
+
unpriced: false
|
|
9632
|
+
};
|
|
9633
|
+
roll.usd += entry.costUsd;
|
|
9634
|
+
roll.entries += 1;
|
|
9635
|
+
if (entry.costUnknown) roll.unpriced = true;
|
|
9636
|
+
perModel.set(entry.model, roll);
|
|
9637
|
+
}
|
|
9638
|
+
return {
|
|
9639
|
+
perChannel: summary.byChannel,
|
|
9640
|
+
total: {
|
|
9641
|
+
usd: summary.totalCostUsd,
|
|
9642
|
+
unknownEntries: summary.byChannel.reduce((sum3, c) => sum3 + c.unpricedCalls, 0)
|
|
9643
|
+
},
|
|
9644
|
+
perModel: [...perModel.values()].sort((a, b) => a.model.localeCompare(b.model))
|
|
9645
|
+
};
|
|
9646
|
+
}
|
|
9647
|
+
function attachCostToReport(report, ledger) {
|
|
9648
|
+
if ("cost" in report) {
|
|
9649
|
+
throw new ValidationError(
|
|
9650
|
+
"attachCostToReport: report already has a 'cost' key \u2014 refusing to overwrite an existing stamp"
|
|
9651
|
+
);
|
|
9652
|
+
}
|
|
9653
|
+
return { ...report, cost: costReport(ledger) };
|
|
9654
|
+
}
|
|
9655
|
+
|
|
9656
|
+
// src/model-seats.ts
|
|
9657
|
+
var seatPresets = {
|
|
9658
|
+
economy: {
|
|
9659
|
+
worker: "kimi-k2.6",
|
|
9660
|
+
judges: ["kimi-k2.6", "deepseek-v4-pro", "gpt-4.1-mini"],
|
|
9661
|
+
analyst: "gpt-4.1-mini",
|
|
9662
|
+
reflection: "gpt-4.1-mini",
|
|
9663
|
+
verifier: "deepseek-v4-pro"
|
|
9664
|
+
},
|
|
9665
|
+
frontier: {}
|
|
9666
|
+
};
|
|
9667
|
+
var SeatUnsetError = class extends ConfigError {
|
|
9668
|
+
constructor(seat) {
|
|
9669
|
+
super(
|
|
9670
|
+
`ModelSeats: seat '${seat}' is unset and no fallback was given \u2014 name a model explicitly (a model id is a budget decision, never a silent default)`
|
|
9671
|
+
);
|
|
9672
|
+
this.seat = seat;
|
|
9673
|
+
}
|
|
9674
|
+
seat;
|
|
9675
|
+
};
|
|
9676
|
+
function resolveSeat(seats, seat, fallback) {
|
|
9677
|
+
const value = seats[seat];
|
|
9678
|
+
if (seat === "judges") {
|
|
9679
|
+
if (value !== void 0 && !Array.isArray(value)) {
|
|
9680
|
+
throw new ValidationError(`ModelSeats: seat 'judges' must be a string[], got ${typeof value}`);
|
|
9681
|
+
}
|
|
9682
|
+
const models = Array.isArray(value) ? value : [];
|
|
9683
|
+
if (models.length > 0) {
|
|
9684
|
+
const blank = models.findIndex((m) => typeof m !== "string" || m.trim() === "");
|
|
9685
|
+
if (blank >= 0) {
|
|
9686
|
+
throw new ValidationError(
|
|
9687
|
+
`ModelSeats: judges[${blank}] is blank \u2014 every panel model must be a non-empty id`
|
|
9688
|
+
);
|
|
9689
|
+
}
|
|
9690
|
+
return [...models];
|
|
9691
|
+
}
|
|
9692
|
+
} else {
|
|
9693
|
+
if (value !== void 0 && typeof value !== "string") {
|
|
9694
|
+
throw new ValidationError(`ModelSeats: seat '${seat}' must be a string, got ${typeof value}`);
|
|
9695
|
+
}
|
|
9696
|
+
if (typeof value === "string" && value.trim() !== "") return value;
|
|
9697
|
+
}
|
|
9698
|
+
if (fallback !== void 0) {
|
|
9699
|
+
if (fallback.trim() === "") {
|
|
9700
|
+
throw new ValidationError(`ModelSeats: fallback for seat '${seat}' is blank`);
|
|
9701
|
+
}
|
|
9702
|
+
return seat === "judges" ? [fallback] : fallback;
|
|
9703
|
+
}
|
|
9704
|
+
throw new SeatUnsetError(seat);
|
|
9705
|
+
}
|
|
9706
|
+
|
|
9707
|
+
// src/verdict-cache.ts
|
|
9708
|
+
import { createHash } from "crypto";
|
|
9709
|
+
import { appendFileSync as appendFileSync3, existsSync as existsSync5, readFileSync as readFileSync6 } from "fs";
|
|
9710
|
+
function canonicalizeAt(value, path) {
|
|
9711
|
+
if (value === null) return "null";
|
|
9712
|
+
switch (typeof value) {
|
|
9713
|
+
case "boolean":
|
|
9714
|
+
return value ? "true" : "false";
|
|
9715
|
+
case "number":
|
|
9716
|
+
if (!Number.isFinite(value)) {
|
|
9717
|
+
throw new Error(
|
|
9718
|
+
`canonicalJson: non-finite number (${value}) at ${path} \u2014 ambiguity is an error, not a coercion`
|
|
9719
|
+
);
|
|
9720
|
+
}
|
|
9721
|
+
return JSON.stringify(value);
|
|
9722
|
+
case "string":
|
|
9723
|
+
return JSON.stringify(value);
|
|
9724
|
+
case "undefined":
|
|
9725
|
+
case "function":
|
|
9726
|
+
case "symbol":
|
|
9727
|
+
throw new Error(
|
|
9728
|
+
`canonicalJson: ${typeof value} at ${path} \u2014 ambiguity is an error, not a coercion`
|
|
9729
|
+
);
|
|
9730
|
+
case "bigint":
|
|
9731
|
+
throw new Error(`canonicalJson: bigint at ${path} \u2014 not representable in JSON`);
|
|
9732
|
+
case "object":
|
|
9733
|
+
break;
|
|
9734
|
+
}
|
|
9735
|
+
const obj = value;
|
|
9736
|
+
if (typeof obj["toJSON"] === "function") {
|
|
9737
|
+
return canonicalizeAt(obj.toJSON(), path);
|
|
9738
|
+
}
|
|
9739
|
+
if (Array.isArray(obj)) {
|
|
9740
|
+
return `[${obj.map((item, i) => canonicalizeAt(item, `${path}[${i}]`)).join(",")}]`;
|
|
9741
|
+
}
|
|
9742
|
+
if (obj instanceof Map || obj instanceof Set) {
|
|
9743
|
+
throw new Error(
|
|
9744
|
+
`canonicalJson: ${obj instanceof Map ? "Map" : "Set"} at ${path} \u2014 would serialize as '{}'; convert to a plain object/array first`
|
|
9745
|
+
);
|
|
9746
|
+
}
|
|
9747
|
+
const keys = Object.keys(obj).sort();
|
|
9748
|
+
const parts = keys.map((k) => `${JSON.stringify(k)}:${canonicalizeAt(obj[k], `${path}.${k}`)}`);
|
|
9749
|
+
return `{${parts.join(",")}}`;
|
|
9750
|
+
}
|
|
9751
|
+
function canonicalJson(value) {
|
|
9752
|
+
return canonicalizeAt(value, "$");
|
|
9753
|
+
}
|
|
9754
|
+
function contentHash(value) {
|
|
9755
|
+
return createHash("sha256").update(canonicalJson(value)).digest("hex");
|
|
9756
|
+
}
|
|
9757
|
+
function inMemoryVerdictCache() {
|
|
9758
|
+
const entries = /* @__PURE__ */ new Map();
|
|
9759
|
+
return {
|
|
9760
|
+
get: (key) => entries.get(key),
|
|
9761
|
+
set: (key, score) => {
|
|
9762
|
+
entries.set(key, score);
|
|
9763
|
+
}
|
|
9764
|
+
};
|
|
9765
|
+
}
|
|
9766
|
+
function parseCacheLine(line, path, lineNo) {
|
|
9767
|
+
let parsed;
|
|
9768
|
+
try {
|
|
9769
|
+
parsed = JSON.parse(line);
|
|
9770
|
+
} catch (err) {
|
|
9771
|
+
throw new Error(
|
|
9772
|
+
`fileVerdictCache: corrupt JSONL at ${path}:${lineNo} \u2014 ${err instanceof Error ? err.message : String(err)}`
|
|
9773
|
+
);
|
|
9774
|
+
}
|
|
9775
|
+
const rec = parsed;
|
|
9776
|
+
if (typeof rec !== "object" || rec === null || typeof rec.key !== "string" || typeof rec.score !== "object" || rec.score === null || typeof rec.score.composite !== "number" || typeof rec.score.dimensions !== "object") {
|
|
9777
|
+
throw new Error(
|
|
9778
|
+
`fileVerdictCache: invalid record shape at ${path}:${lineNo} \u2014 expected {key, score:{dimensions, composite, notes}}`
|
|
9779
|
+
);
|
|
9780
|
+
}
|
|
9781
|
+
return rec;
|
|
9782
|
+
}
|
|
9783
|
+
function fileVerdictCache(path) {
|
|
9784
|
+
const entries = /* @__PURE__ */ new Map();
|
|
9785
|
+
if (existsSync5(path)) {
|
|
9786
|
+
const lines = readFileSync6(path, "utf8").split("\n");
|
|
9787
|
+
for (let i = 0; i < lines.length; i++) {
|
|
9788
|
+
const line = lines[i];
|
|
9789
|
+
if (line === void 0 || line.trim() === "") continue;
|
|
9790
|
+
const rec = parseCacheLine(line, path, i + 1);
|
|
9791
|
+
entries.set(rec.key, rec.score);
|
|
9792
|
+
}
|
|
9793
|
+
}
|
|
9794
|
+
return {
|
|
9795
|
+
get: (key) => entries.get(key),
|
|
9796
|
+
set: (key, score) => {
|
|
9797
|
+
appendFileSync3(path, `${JSON.stringify({ key, score })}
|
|
9798
|
+
`, "utf8");
|
|
9799
|
+
entries.set(key, score);
|
|
9800
|
+
}
|
|
9801
|
+
};
|
|
9802
|
+
}
|
|
9803
|
+
function cachedJudge(judge, store, options) {
|
|
9804
|
+
if (typeof options.judgeVersion !== "string" || options.judgeVersion.trim() === "") {
|
|
9805
|
+
throw new Error("cachedJudge: judgeVersion is required and must be a non-empty string");
|
|
9806
|
+
}
|
|
9807
|
+
const stats = { hits: 0, misses: 0 };
|
|
9808
|
+
const wrapped = {
|
|
9809
|
+
name: judge.name,
|
|
9810
|
+
dimensions: judge.dimensions,
|
|
9811
|
+
async score(input) {
|
|
9812
|
+
const key = contentHash({
|
|
9813
|
+
artifact: canonicalJson(input.artifact),
|
|
9814
|
+
scenarioId: input.scenario.id,
|
|
9815
|
+
judgeName: judge.name,
|
|
9816
|
+
dimensions: judge.dimensions,
|
|
9817
|
+
judgeVersion: options.judgeVersion
|
|
9818
|
+
});
|
|
9819
|
+
const cached = await store.get(key);
|
|
9820
|
+
if (cached !== void 0) {
|
|
9821
|
+
stats.hits += 1;
|
|
9822
|
+
return cached;
|
|
9823
|
+
}
|
|
9824
|
+
const score = await judge.score(input);
|
|
9825
|
+
await store.set(key, score);
|
|
9826
|
+
stats.misses += 1;
|
|
9827
|
+
return score;
|
|
9828
|
+
},
|
|
9829
|
+
stats: () => ({ ...stats })
|
|
9830
|
+
};
|
|
9831
|
+
if (judge.appliesTo) wrapped.appliesTo = judge.appliesTo;
|
|
9832
|
+
return wrapped;
|
|
9833
|
+
}
|
|
9834
|
+
|
|
9835
|
+
// src/attestation.ts
|
|
9836
|
+
var ATTESTATION_ALGORITHM = "sha256/canonical-json";
|
|
9837
|
+
function attest(report, provenance) {
|
|
9838
|
+
return {
|
|
9839
|
+
reportHash: contentHash(report),
|
|
9840
|
+
provenance,
|
|
9841
|
+
algorithm: ATTESTATION_ALGORITHM
|
|
9842
|
+
};
|
|
9843
|
+
}
|
|
9844
|
+
function verifyAttestation(report, attested) {
|
|
9845
|
+
if (attested.algorithm !== ATTESTATION_ALGORITHM) {
|
|
9846
|
+
return {
|
|
9847
|
+
valid: false,
|
|
9848
|
+
reason: `unknown algorithm '${attested.algorithm}' \u2014 this verifier only checks '${ATTESTATION_ALGORITHM}'`
|
|
9849
|
+
};
|
|
9850
|
+
}
|
|
9851
|
+
let recomputed;
|
|
9852
|
+
try {
|
|
9853
|
+
recomputed = contentHash(report);
|
|
9854
|
+
} catch (err) {
|
|
9855
|
+
return {
|
|
9856
|
+
valid: false,
|
|
9857
|
+
reason: `report is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
|
|
9858
|
+
};
|
|
9859
|
+
}
|
|
9860
|
+
if (recomputed !== attested.reportHash) {
|
|
9861
|
+
return {
|
|
9862
|
+
valid: false,
|
|
9863
|
+
reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
|
|
9864
|
+
};
|
|
9865
|
+
}
|
|
9866
|
+
return { valid: true };
|
|
9867
|
+
}
|
|
9097
9868
|
export {
|
|
9098
9869
|
AGENT_PROFILE_KINDS,
|
|
9870
|
+
ATTESTATION_ALGORITHM,
|
|
9099
9871
|
AgentDriver,
|
|
9100
9872
|
AgentEvalError,
|
|
9101
9873
|
AgentProfileCellValidationError,
|
|
@@ -9150,6 +9922,7 @@ export {
|
|
|
9150
9922
|
InMemoryTraceStore,
|
|
9151
9923
|
InMemoryWorkspaceInspector,
|
|
9152
9924
|
JudgeError,
|
|
9925
|
+
JudgeParseError,
|
|
9153
9926
|
JudgeRunner,
|
|
9154
9927
|
KNOWLEDGE_GAP_KIND_SPEC,
|
|
9155
9928
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
@@ -9182,6 +9955,7 @@ export {
|
|
|
9182
9955
|
SKILL_USAGE_ANALYST,
|
|
9183
9956
|
SandboxHarness,
|
|
9184
9957
|
ScenarioRegistry,
|
|
9958
|
+
SeatUnsetError,
|
|
9185
9959
|
SingleBackendError,
|
|
9186
9960
|
SkillUsageAnalyst,
|
|
9187
9961
|
SpanNotFoundError,
|
|
@@ -9192,6 +9966,7 @@ export {
|
|
|
9192
9966
|
TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
|
|
9193
9967
|
TRACE_SCHEMA_VERSION,
|
|
9194
9968
|
TokenCounter,
|
|
9969
|
+
TraceContractBuilder,
|
|
9195
9970
|
TraceEmitter,
|
|
9196
9971
|
TraceFileMissingError,
|
|
9197
9972
|
TraceNotFoundError,
|
|
@@ -9227,6 +10002,8 @@ export {
|
|
|
9227
10002
|
assertSingleBackend,
|
|
9228
10003
|
assignFeedbackSplit,
|
|
9229
10004
|
assignHeldOutTag,
|
|
10005
|
+
attachCostToReport,
|
|
10006
|
+
attest,
|
|
9230
10007
|
attributeCounterfactuals,
|
|
9231
10008
|
backoffMs,
|
|
9232
10009
|
deterministicSplit as benchmarkDeterministicSplit,
|
|
@@ -9248,17 +10025,20 @@ export {
|
|
|
9248
10025
|
buildTraceInsightPrompt,
|
|
9249
10026
|
buildTrajectory,
|
|
9250
10027
|
byteLengthRange,
|
|
10028
|
+
cachedJudge,
|
|
9251
10029
|
calibrateJudge,
|
|
9252
10030
|
calibrateJudgeContinuous,
|
|
9253
10031
|
callLlm,
|
|
9254
10032
|
callLlmJson,
|
|
9255
10033
|
canaryLeakView,
|
|
10034
|
+
canonicalJson,
|
|
9256
10035
|
canonicalize,
|
|
9257
10036
|
captureFetchToRawSink,
|
|
9258
10037
|
causalAttribution,
|
|
9259
10038
|
checkBehavioralCanary,
|
|
9260
10039
|
checkCanaries,
|
|
9261
10040
|
checkSlos,
|
|
10041
|
+
checkTraceContracts,
|
|
9262
10042
|
clamp01,
|
|
9263
10043
|
classifyEuAiRisk,
|
|
9264
10044
|
classifyFailure,
|
|
@@ -9272,6 +10052,7 @@ export {
|
|
|
9272
10052
|
compareReferenceReplay,
|
|
9273
10053
|
compareToBaseline,
|
|
9274
10054
|
compilerJudge,
|
|
10055
|
+
completionVerdict,
|
|
9275
10056
|
composeParsers,
|
|
9276
10057
|
composeValidators,
|
|
9277
10058
|
computeExperimentStats,
|
|
@@ -9280,7 +10061,9 @@ export {
|
|
|
9280
10061
|
computeTraceMetrics,
|
|
9281
10062
|
confidenceInterval,
|
|
9282
10063
|
containsAll,
|
|
10064
|
+
contentHash,
|
|
9283
10065
|
continuousAgreement,
|
|
10066
|
+
contractJudge,
|
|
9284
10067
|
controlFailureClassFromVerification,
|
|
9285
10068
|
controlRunToFeedbackTrajectory,
|
|
9286
10069
|
controlRunToRunRecord,
|
|
@@ -9288,6 +10071,7 @@ export {
|
|
|
9288
10071
|
corpusInterRaterAgreement,
|
|
9289
10072
|
corpusInterRaterAgreementFromJudgeScores,
|
|
9290
10073
|
costForUsage,
|
|
10074
|
+
costReport,
|
|
9291
10075
|
createAnalystAi,
|
|
9292
10076
|
createAntiSlopJudge,
|
|
9293
10077
|
createChatClient,
|
|
@@ -9326,6 +10110,8 @@ export {
|
|
|
9326
10110
|
distillPlaybook,
|
|
9327
10111
|
domainEvidencePattern,
|
|
9328
10112
|
dominates,
|
|
10113
|
+
eProcess,
|
|
10114
|
+
ensembleJudge,
|
|
9329
10115
|
estimateCost,
|
|
9330
10116
|
estimateTokens,
|
|
9331
10117
|
euAiActReport,
|
|
@@ -9335,6 +10121,7 @@ export {
|
|
|
9335
10121
|
evaluateInterimReleaseConfidence,
|
|
9336
10122
|
evaluateOracles,
|
|
9337
10123
|
evaluateReleaseConfidence,
|
|
10124
|
+
evaluateTraceContract,
|
|
9338
10125
|
executeScenario,
|
|
9339
10126
|
expectAgent,
|
|
9340
10127
|
exportRewardModel,
|
|
@@ -9354,6 +10141,7 @@ export {
|
|
|
9354
10141
|
fileContains,
|
|
9355
10142
|
fileExists,
|
|
9356
10143
|
fileExperimentStore,
|
|
10144
|
+
fileVerdictCache,
|
|
9357
10145
|
findAutoMatchNoExpectation,
|
|
9358
10146
|
findConstructorCwdDropped,
|
|
9359
10147
|
findFallbackToPass,
|
|
@@ -9386,6 +10174,7 @@ export {
|
|
|
9386
10174
|
inMemoryReferenceReplayStore,
|
|
9387
10175
|
inMemoryReviewStore,
|
|
9388
10176
|
inMemoryRunRecordBackend,
|
|
10177
|
+
inMemoryVerdictCache,
|
|
9389
10178
|
inferDomainKeywords,
|
|
9390
10179
|
inferOtlpKind,
|
|
9391
10180
|
interRaterReliability,
|
|
@@ -9420,13 +10209,16 @@ export {
|
|
|
9420
10209
|
loadScorerFromGrader,
|
|
9421
10210
|
localCommandRunner,
|
|
9422
10211
|
lowercaseMutator,
|
|
10212
|
+
makeEvalTools,
|
|
9423
10213
|
makeFinding,
|
|
9424
10214
|
mannWhitneyU,
|
|
9425
10215
|
matchGoldens,
|
|
10216
|
+
matchSpan,
|
|
9426
10217
|
mergeLayerResults,
|
|
9427
10218
|
mergeSteeringBundle,
|
|
9428
10219
|
modelDescriptionBits,
|
|
9429
10220
|
modelPriceKey,
|
|
10221
|
+
mulberry32,
|
|
9430
10222
|
multiToolchainLayer,
|
|
9431
10223
|
nistAiRmfReport,
|
|
9432
10224
|
normalizeScores,
|
|
@@ -9495,6 +10287,7 @@ export {
|
|
|
9495
10287
|
researchReport,
|
|
9496
10288
|
resetLockedAppendersForTesting,
|
|
9497
10289
|
resolveModelPricing,
|
|
10290
|
+
resolveSeat,
|
|
9498
10291
|
roundTripRunRecord,
|
|
9499
10292
|
rowCount,
|
|
9500
10293
|
rowWhere,
|
|
@@ -9532,6 +10325,7 @@ export {
|
|
|
9532
10325
|
scoreRedTeamOutput,
|
|
9533
10326
|
scoreReferenceReplay,
|
|
9534
10327
|
scoreTraceInsightReadiness,
|
|
10328
|
+
seatPresets,
|
|
9535
10329
|
securityJudge,
|
|
9536
10330
|
selectHarnessVariant,
|
|
9537
10331
|
selfPreference,
|
|
@@ -9557,12 +10351,14 @@ export {
|
|
|
9557
10351
|
throwIfRunIncomplete,
|
|
9558
10352
|
toAgentProfileJson,
|
|
9559
10353
|
toLangfuseEnvelope,
|
|
10354
|
+
toOpenAiTool,
|
|
9560
10355
|
toPrometheusText,
|
|
9561
10356
|
tokenizeDomainWords,
|
|
9562
10357
|
toolNamesForRun,
|
|
9563
10358
|
toolSpans,
|
|
9564
10359
|
traceAnalystFunctionGroup,
|
|
9565
10360
|
traceAnalystOnRunComplete,
|
|
10361
|
+
traceContract,
|
|
9566
10362
|
traceJudge,
|
|
9567
10363
|
traceJudgeEnsemble,
|
|
9568
10364
|
tracedAnalyzeTraces,
|
|
@@ -9573,6 +10369,7 @@ export {
|
|
|
9573
10369
|
validateRunRecord,
|
|
9574
10370
|
verbosityBias,
|
|
9575
10371
|
verifyAgentProfileCell,
|
|
10372
|
+
verifyAttestation,
|
|
9576
10373
|
verifyCompletion,
|
|
9577
10374
|
verifyManifest,
|
|
9578
10375
|
visualDiff,
|