@tangle-network/agent-eval 0.86.0 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/dist/adapters/http.d.ts +3 -3
  2. package/dist/adapters/langchain.d.ts +3 -3
  3. package/dist/adapters/otel.d.ts +6 -6
  4. package/dist/adversarial-DIVcDoI_.d.ts +88 -0
  5. package/dist/analyst/index.d.ts +11 -10
  6. package/dist/analyst/index.js +13 -8
  7. package/dist/analyst/index.js.map +1 -1
  8. package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
  9. package/dist/belief-state/index.d.ts +4 -4
  10. package/dist/belief-state/index.js +1 -1
  11. package/dist/benchmarks/index.d.ts +3 -3
  12. package/dist/campaign/index.d.ts +165 -18
  13. package/dist/campaign/index.js +289 -14
  14. package/dist/campaign/index.js.map +1 -1
  15. package/dist/chunk-45EEMHTC.js +35 -0
  16. package/dist/chunk-45EEMHTC.js.map +1 -0
  17. package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
  18. package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
  19. package/dist/chunk-5HRORJQY.js.map +1 -0
  20. package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
  21. package/dist/chunk-BOD4O7OF.js +40 -0
  22. package/dist/chunk-BOD4O7OF.js.map +1 -0
  23. package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
  24. package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
  25. package/dist/chunk-D3V5B42D.js.map +1 -0
  26. package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
  27. package/dist/chunk-FIUKOSWI.js.map +1 -0
  28. package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
  29. package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
  30. package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
  31. package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
  32. package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
  33. package/dist/chunk-QAY5UIJO.js.map +1 -0
  34. package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
  35. package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
  36. package/dist/chunk-REVYNR6C.js +100 -0
  37. package/dist/chunk-REVYNR6C.js.map +1 -0
  38. package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
  39. package/dist/chunk-UHMJT4T7.js +200 -0
  40. package/dist/chunk-UHMJT4T7.js.map +1 -0
  41. package/dist/chunk-UMMZHCPB.js +190 -0
  42. package/dist/chunk-UMMZHCPB.js.map +1 -0
  43. package/dist/chunk-VZSRQ272.js +149 -0
  44. package/dist/chunk-VZSRQ272.js.map +1 -0
  45. package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
  46. package/dist/chunk-XY4DDNEG.js.map +1 -0
  47. package/dist/chunk-Y47J2LJ3.js +859 -0
  48. package/dist/chunk-Y47J2LJ3.js.map +1 -0
  49. package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
  50. package/dist/chunk-ZFIBGEOL.js.map +1 -0
  51. package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
  52. package/dist/contract/index.d.ts +24 -95
  53. package/dist/contract/index.js +16 -755
  54. package/dist/contract/index.js.map +1 -1
  55. package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
  56. package/dist/control.d.ts +5 -5
  57. package/dist/corpus-BoR-041R.d.ts +560 -0
  58. package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
  59. package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
  60. package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
  61. package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
  62. package/dist/diagnose.d.ts +251 -0
  63. package/dist/diagnose.js +381 -0
  64. package/dist/diagnose.js.map +1 -0
  65. package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
  66. package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
  67. package/dist/fuzz.d.ts +484 -0
  68. package/dist/fuzz.js +613 -0
  69. package/dist/fuzz.js.map +1 -0
  70. package/dist/governance/index.d.ts +4 -4
  71. package/dist/hosted/index.d.ts +6 -6
  72. package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
  73. package/dist/index.d.ts +717 -455
  74. package/dist/index.js +1590 -793
  75. package/dist/index.js.map +1 -1
  76. package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
  77. package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
  78. package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
  79. package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
  80. package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
  81. package/dist/matrix/index.d.ts +2 -2
  82. package/dist/meta-eval/index.d.ts +177 -3
  83. package/dist/meta-eval/index.js +260 -1
  84. package/dist/meta-eval/index.js.map +1 -1
  85. package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
  86. package/dist/multishot/index.d.ts +25 -11
  87. package/dist/multishot/index.js +36 -7
  88. package/dist/multishot/index.js.map +1 -1
  89. package/dist/openapi.json +1 -1
  90. package/dist/pipelines/index.js +2 -2
  91. package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
  92. package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
  93. package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
  94. package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
  95. package/dist/reporting.d.ts +8 -8
  96. package/dist/reporting.js +3 -3
  97. package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
  98. package/dist/rl.d.ts +194 -656
  99. package/dist/rl.js +236 -154
  100. package/dist/rl.js.map +1 -1
  101. package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
  102. package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
  103. package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
  104. package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
  105. package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
  106. package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
  107. package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
  108. package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
  109. package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
  110. package/dist/traces.d.ts +3 -3
  111. package/dist/traces.js +8 -6
  112. package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
  113. package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
  114. package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
  115. package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
  116. package/dist/wire/index.d.ts +3 -3
  117. package/dist/workflow/index.d.ts +12 -11
  118. package/dist/workflow/index.js +1 -1
  119. package/package.json +11 -1
  120. package/dist/chunk-BABOZOSN.js.map +0 -1
  121. package/dist/chunk-L5G7OUKD.js.map +0 -1
  122. package/dist/chunk-SHTXZ4O2.js +0 -113
  123. package/dist/chunk-SHTXZ4O2.js.map +0 -1
  124. package/dist/chunk-VIDQF3F5.js.map +0 -1
  125. package/dist/chunk-VUINJM5M.js.map +0 -1
  126. package/dist/chunk-YGYXHNAQ.js.map +0 -1
  127. package/dist/chunk-YV7J7X5N.js.map +0 -1
  128. /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
  129. /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
  130. /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
  131. /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
  132. /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
  133. /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
  134. /package/dist/{run-campaign-4Y5V5CN3.js.map → run-campaign-RDGAM5KJ.js.map} +0 -0
package/dist/index.js CHANGED
@@ -1,22 +1,35 @@
1
+ import {
2
+ HoldoutAuditor,
3
+ analyzeRuns,
4
+ canaryLeakView,
5
+ checkBehavioralCanary,
6
+ checkCanaries,
7
+ runBehavioralCanaries
8
+ } from "./chunk-Y47J2LJ3.js";
9
+ import {
10
+ classifyEuAiRisk,
11
+ euAiActReport,
12
+ nistAiRmfReport,
13
+ renderMarkdown,
14
+ soc2Report,
15
+ summarize
16
+ } from "./chunk-KKHDIONI.js";
17
+ import {
18
+ acquisitionPlansForKnowledgeGaps,
19
+ blockingKnowledgeEval,
20
+ knowledgeReadinessTracePayload,
21
+ scoreKnowledgeReadiness,
22
+ userQuestionsForKnowledgeGaps
23
+ } from "./chunk-3CKU6VGU.js";
1
24
  import {
2
25
  agentProfileHash,
26
+ completionVerdict,
3
27
  createLlmCorrectnessChecker,
4
28
  createTokenRecallChecker,
5
29
  extractProducedState,
6
30
  parseCorrectnessResponse,
7
31
  verifyCompletion
8
- } from "./chunk-YGYXHNAQ.js";
9
- import {
10
- parseRuntimeTrajectoryHookEvent,
11
- projectRuntimeTrajectoryEvidence
12
- } from "./chunk-T4SQEITX.js";
13
- import {
14
- HoldoutAuditor,
15
- canaryLeakView,
16
- checkBehavioralCanary,
17
- checkCanaries,
18
- runBehavioralCanaries
19
- } from "./chunk-SHTXZ4O2.js";
32
+ } from "./chunk-FIUKOSWI.js";
20
33
  import {
21
34
  DEFAULT_MUTATION_PRIMITIVES,
22
35
  DEFAULT_RED_TEAM_CORPUS,
@@ -40,12 +53,16 @@ import {
40
53
  scoreRedTeamOutput,
41
54
  surfaceContentHash,
42
55
  toolNamesForRun
43
- } from "./chunk-BABOZOSN.js";
56
+ } from "./chunk-ZFIBGEOL.js";
44
57
  import {
45
58
  BackendIntegrityError,
46
59
  assertRealBackend,
47
60
  summarizeBackendIntegrity
48
- } from "./chunk-ZZ2HOPME.js";
61
+ } from "./chunk-TWS7AZEY.js";
62
+ import {
63
+ parseRuntimeTrajectoryHookEvent,
64
+ projectRuntimeTrajectoryEvidence
65
+ } from "./chunk-T4SQEITX.js";
49
66
  import {
50
67
  MODEL_PRICING,
51
68
  MetricsCollector,
@@ -67,14 +84,14 @@ import {
67
84
  computeToolUseMetrics,
68
85
  iqr,
69
86
  welchsTTest
70
- } from "./chunk-RBNA5AZT.js";
87
+ } from "./chunk-L3JOU6XM.js";
88
+ import {
89
+ analyzeSeries
90
+ } from "./chunk-BOD4O7OF.js";
71
91
  import {
72
92
  exportTrainingData,
73
93
  toNdjson
74
94
  } from "./chunk-KMPRBJK4.js";
75
- import {
76
- buildTrajectory
77
- } from "./chunk-RZTMDUO7.js";
78
95
  import {
79
96
  DockerSandboxDriver,
80
97
  SandboxHarness,
@@ -85,21 +102,6 @@ import {
85
102
  runTestGradedScenario,
86
103
  vitestTestParser
87
104
  } from "./chunk-T375SUOZ.js";
88
- import {
89
- classifyEuAiRisk,
90
- euAiActReport,
91
- nistAiRmfReport,
92
- renderMarkdown,
93
- soc2Report,
94
- summarize
95
- } from "./chunk-KKHDIONI.js";
96
- import {
97
- acquisitionPlansForKnowledgeGaps,
98
- blockingKnowledgeEval,
99
- knowledgeReadinessTracePayload,
100
- scoreKnowledgeReadiness,
101
- userQuestionsForKnowledgeGaps
102
- } from "./chunk-3CKU6VGU.js";
103
105
  import {
104
106
  DEFAULT_COMPLEXITY_WEIGHTS,
105
107
  DEFAULT_RUN_SCORE_WEIGHTS,
@@ -111,9 +113,7 @@ import {
111
113
  SKILL_USAGE_ANALYST,
112
114
  SkillUsageAnalyst,
113
115
  aggregateRunScore,
114
- buildDefaultAnalystRegistry,
115
116
  clamp01,
116
- computeTraceMetrics,
117
117
  createAnalystAi,
118
118
  createChatClient,
119
119
  createSemanticConceptJudge,
@@ -121,7 +121,11 @@ import {
121
121
  diffFindings,
122
122
  resetLockedAppendersForTesting,
123
123
  runSemanticConceptJudge
124
- } from "./chunk-L5G7OUKD.js";
124
+ } from "./chunk-XY4DDNEG.js";
125
+ import {
126
+ buildDefaultAnalystRegistry,
127
+ computeTraceMetrics
128
+ } from "./chunk-UMMZHCPB.js";
125
129
  import {
126
130
  AnalystRegistry,
127
131
  DEFAULT_TRACE_ANALYST_KINDS,
@@ -129,11 +133,9 @@ import {
129
133
  IMPROVEMENT_KIND_SPEC,
130
134
  KNOWLEDGE_GAP_KIND_SPEC,
131
135
  KNOWLEDGE_POISONING_KIND_SPEC,
132
- computeFindingId,
133
136
  createTraceAnalystKind,
134
- makeFinding,
135
137
  renderPriorFindings
136
- } from "./chunk-VIDQF3F5.js";
138
+ } from "./chunk-D3V5B42D.js";
137
139
  import {
138
140
  controlFailureClassFromVerification,
139
141
  controlRunToRunRecord,
@@ -159,10 +161,10 @@ import {
159
161
  evaluateReleaseConfidence,
160
162
  judgeReplayGate,
161
163
  renderReleaseReport
162
- } from "./chunk-FZWAFVAA.js";
164
+ } from "./chunk-4FBZZIYD.js";
163
165
  import {
164
166
  runEvalCampaign
165
- } from "./chunk-WJL2NJXN.js";
167
+ } from "./chunk-GSH6QNNS.js";
166
168
  import {
167
169
  LlmCallError,
168
170
  LlmClient,
@@ -185,7 +187,18 @@ import {
185
187
  paretoChart,
186
188
  researchReport,
187
189
  summaryTable
188
- } from "./chunk-Z7VFTS2J.js";
190
+ } from "./chunk-CY6U5S3X.js";
191
+ import {
192
+ attributeCounterfactuals,
193
+ runCounterfactual
194
+ } from "./chunk-REVYNR6C.js";
195
+ import {
196
+ buildTrajectory
197
+ } from "./chunk-RZTMDUO7.js";
198
+ import {
199
+ computeFindingId,
200
+ makeFinding
201
+ } from "./chunk-45EEMHTC.js";
189
202
  import {
190
203
  benjaminiHochberg,
191
204
  bonferroni,
@@ -197,9 +210,11 @@ import {
197
210
  continuousAgreement,
198
211
  corpusInterRaterAgreement,
199
212
  corpusInterRaterAgreementFromJudgeScores,
213
+ eProcess,
200
214
  interRaterReliability,
201
215
  interpretCliffs,
202
216
  mannWhitneyU,
217
+ mulberry32,
203
218
  normalizeScores,
204
219
  pairedBootstrap,
205
220
  pairedMde,
@@ -212,7 +227,7 @@ import {
212
227
  weightedComposite,
213
228
  weightedMean,
214
229
  wilcoxonSignedRank
215
- } from "./chunk-IDVBLYCY.js";
230
+ } from "./chunk-LMZQ2Z4U.js";
216
231
  import {
217
232
  FileSystemTraceStore,
218
233
  InMemoryTraceStore,
@@ -243,7 +258,7 @@ import {
243
258
  scoreTraceInsightReadiness,
244
259
  tokenizeDomainWords,
245
260
  traceAnalystOnRunComplete
246
- } from "./chunk-P2J6SOXT.js";
261
+ } from "./chunk-QG2OVF2D.js";
247
262
  import {
248
263
  DEFAULT_REDACTION_RULES,
249
264
  REDACTION_VERSION,
@@ -270,16 +285,18 @@ import {
270
285
  isToolSpan
271
286
  } from "./chunk-5BKGXME7.js";
272
287
  import {
273
- DEFAULT_TRACE_ANALYST_BUDGETS,
274
- OtlpFileTraceStore,
275
- SpanNotFoundError,
276
288
  TRACE_ANALYST_ACTOR_DESCRIPTION,
277
289
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
278
290
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
291
+ analyzeTraces
292
+ } from "./chunk-UHMJT4T7.js";
293
+ import {
294
+ DEFAULT_TRACE_ANALYST_BUDGETS,
295
+ OtlpFileTraceStore,
296
+ SpanNotFoundError,
279
297
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
280
298
  TraceFileMissingError,
281
299
  TraceNotFoundError,
282
- analyzeTraces,
283
300
  asNumber,
284
301
  asString,
285
302
  buildTraceAnalystTools,
@@ -291,7 +308,7 @@ import {
291
308
  readOtlpStatus,
292
309
  stringField,
293
310
  traceAnalystFunctionGroup
294
- } from "./chunk-VUINJM5M.js";
311
+ } from "./chunk-QAY5UIJO.js";
295
312
  import {
296
313
  RunIntegrityError,
297
314
  assertRunCaptured,
@@ -646,132 +663,337 @@ function renderCommitMessage(input) {
646
663
  return lines.join("\n").trim();
647
664
  }
648
665
 
649
- // src/executor.ts
650
- async function executeScenario(tc, scenario, config) {
651
- const startTime = Date.now();
652
- const model = config.model ?? "gpt-4o";
653
- const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
654
- const messages = [{ role: "system", content: systemPrompt }];
655
- const turns = [];
656
- const allCodeBlocks = [];
657
- const allBlocks = [];
658
- const allToolCalls = [];
659
- const blockRe = config.blockPattern ?? /:::(\w+)\s*\n([\s\S]*?)\n\s*:::/g;
660
- for (let i = 0; i < scenario.turns.length; i++) {
661
- const turn = scenario.turns[i];
662
- const turnStart = Date.now();
663
- messages.push({ role: "user", content: turn.user });
666
+ // src/judges.ts
667
+ var JudgeParseError = class extends JudgeError {
668
+ /** Name of the judge whose response failed to parse. */
669
+ judgeName;
670
+ /** The raw (truncated) model response that failed to parse. */
671
+ raw;
672
+ constructor(judgeName, raw, options) {
673
+ super(`judge '${judgeName}' returned an unparseable response: ${raw.slice(0, 200)}`, options);
674
+ this.judgeName = judgeName;
675
+ this.raw = raw;
676
+ }
677
+ };
678
+ function createDomainExpertJudge(domain) {
679
+ return async (tc, { scenario, turns }) => {
680
+ const conversation = turns.map(
681
+ (t, i) => `Turn ${i + 1}:
682
+ User: ${t.userMessage}
683
+ Agent: ${t.agentResponse.slice(0, 2e3)}`
684
+ ).join("\n\n---\n\n");
664
685
  const resp = await tc.chat({
665
- model,
666
- messages,
667
- temperature: 0.4,
668
- maxTokens: 3e3
669
- });
670
- const content = resp.choices?.[0]?.message?.content ?? "";
671
- messages.push({ role: "assistant", content });
672
- const codeRe = /```(\w+)?\n([\s\S]*?)```/g;
673
- let codeMatch = codeRe.exec(content);
674
- while (codeMatch !== null) {
675
- allCodeBlocks.push({ language: codeMatch[1] ?? "text", code: codeMatch[2] ?? "" });
676
- codeMatch = codeRe.exec(content);
677
- }
678
- const turnBlocks = [];
679
- const blockReLocal = new RegExp(blockRe.source, blockRe.flags);
680
- let blockMatch = blockReLocal.exec(content);
681
- while (blockMatch !== null) {
682
- const fields = {};
683
- for (const line of (blockMatch[2] ?? "").split("\n")) {
684
- const idx = line.indexOf(":");
685
- if (idx > 0) fields[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
686
- }
687
- const blockType = blockMatch[1] ?? "";
688
- allBlocks.push({ type: blockType, fields });
689
- turnBlocks.push({ type: blockType, title: fields.title ?? "" });
690
- blockMatch = blockReLocal.exec(content);
691
- }
692
- let hasToolCall = false;
693
- if (config.toolCallPatterns) {
694
- for (const pattern of config.toolCallPatterns) {
695
- const re = new RegExp(pattern.source, pattern.flags);
696
- let toolMatch = re.exec(content);
697
- while (toolMatch !== null) {
698
- allToolCalls.push(toolMatch[0]);
699
- hasToolCall = true;
700
- toolMatch = re.exec(content);
686
+ model: "gpt-4o",
687
+ messages: [
688
+ {
689
+ role: "system",
690
+ content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
691
+
692
+ Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
693
+
694
+ Evaluate:
695
+ 1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
696
+ 2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
697
+
698
+ Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
699
+ },
700
+ {
701
+ role: "user",
702
+ content: `Persona: ${scenario.persona} (${scenario.label})
703
+ Scenario: ${scenario.thesis}
704
+
705
+ ${conversation}`
701
706
  }
702
- }
703
- }
704
- turns.push({
705
- turnIndex: i,
706
- userMessage: turn.user,
707
- agentResponse: content,
708
- durationMs: Date.now() - turnStart,
709
- blocksExtracted: turnBlocks,
710
- containsCode: allCodeBlocks.length > 0,
711
- containsToolCall: hasToolCall
707
+ ],
708
+ temperature: 0.1,
709
+ maxTokens: 800
712
710
  });
713
- }
714
- const artifacts = {
715
- vaultFiles: [],
716
- blocksExtracted: allBlocks,
717
- codeBlocks: allCodeBlocks,
718
- toolCalls: allToolCalls
711
+ return parseJudgeResponse("domain_expert", resp);
719
712
  };
720
- const artifactResults = scenario.artifactChecks.map((check2) => {
721
- if (config.artifactChecker) {
722
- const custom = config.artifactChecker(check2, artifacts);
723
- if (custom) return { check: check2, ...custom };
724
- }
725
- switch (check2.type) {
726
- case "block_extracted": {
727
- const count = allBlocks.filter((b) => b.type === check2.target).length;
728
- return {
729
- check: check2,
730
- passed: count >= (check2.minCount ?? 1),
731
- detail: `Found ${count} ${check2.target} blocks (need ${check2.minCount ?? 1})`
732
- };
713
+ }
714
+ var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
715
+ const codeBlocks = artifacts.codeBlocks;
716
+ if (codeBlocks.length === 0) {
717
+ return [
718
+ {
719
+ judgeName: "code_execution",
720
+ dimension: "code_execution",
721
+ score: 0,
722
+ reasoning: "No code blocks found in agent response."
733
723
  }
734
- case "code_valid": {
735
- const hasCode = allCodeBlocks.some(
736
- (b) => b.language === check2.target || b.code.includes(check2.target)
737
- );
738
- return { check: check2, passed: hasCode, detail: hasCode ? "Code block found" : "No matching code" };
724
+ ];
725
+ }
726
+ const codeText = codeBlocks.map(
727
+ (b, i) => `Block ${i + 1} (${b.language}):
728
+ \`\`\`${b.language}
729
+ ${b.code.slice(0, 3e3)}
730
+ \`\`\``
731
+ ).join("\n\n");
732
+ const resp = await tc.chat({
733
+ model: "gpt-4o",
734
+ messages: [
735
+ {
736
+ role: "system",
737
+ content: `You are a principal software engineer reviewing code written by an AI agent.
738
+
739
+ Score STRICTLY:
740
+ 1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
741
+ 2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
742
+ 3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
743
+
744
+ Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
745
+ },
746
+ {
747
+ role: "user",
748
+ content: `Task: ${scenario.thesis}
749
+
750
+ ${codeText}`
739
751
  }
740
- default:
741
- return {
742
- check: check2,
743
- passed: false,
744
- detail: `Check type "${check2.type}" requires live environment`
745
- };
746
- }
752
+ ],
753
+ temperature: 0.1,
754
+ maxTokens: 1e3
747
755
  });
748
- const judgeInput = { scenario, turns, artifacts };
749
- const judgeResults = [];
750
- for (const judge of config.judges) {
751
- let lastErr = "";
752
- for (let attempt = 0; attempt < 3; attempt++) {
753
- try {
754
- if (attempt > 0) {
755
- const wait = attempt * 1e4;
756
- console.log(` judge retry ${attempt}/2 (waiting ${wait / 1e3}s)`);
757
- await new Promise((r) => setTimeout(r, wait));
758
- }
759
- const scores2 = await judge(tc, judgeInput);
760
- judgeResults.push(scores2);
761
- await new Promise((r) => setTimeout(r, 3e3));
762
- break;
763
- } catch (err) {
764
- lastErr = err instanceof Error ? err.message : String(err);
765
- if (attempt === 2) {
766
- judgeResults.push([
767
- {
768
- judgeName: "unknown",
769
- dimension: "error",
770
- score: 0,
771
- reasoning: `Judge failed after 3 attempts: ${lastErr.slice(0, 200)}`
772
- }
773
- ]);
756
+ return parseJudgeResponse("code_execution", resp);
757
+ };
758
+ var coherenceJudge = async (tc, { scenario, turns }) => {
759
+ if (turns.length < 2) {
760
+ return [];
761
+ }
762
+ const conversation = turns.map(
763
+ (t, i) => `Turn ${i + 1}:
764
+ User: ${t.userMessage}
765
+ Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
766
+ ).join("\n\n---\n\n");
767
+ const resp = await tc.chat({
768
+ model: "gpt-4o",
769
+ messages: [
770
+ {
771
+ role: "system",
772
+ content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
773
+
774
+ Score STRICTLY:
775
+ 1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
776
+ 2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
777
+ 3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
778
+
779
+ Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
780
+ },
781
+ {
782
+ role: "user",
783
+ content: `Scenario: ${scenario.thesis}
784
+
785
+ ${conversation}`
786
+ }
787
+ ],
788
+ temperature: 0.1,
789
+ maxTokens: 800
790
+ });
791
+ return parseJudgeResponse("coherence", resp);
792
+ };
793
+ var adversarialJudge = async (tc, { scenario, turns }) => {
794
+ const conversation = turns.map(
795
+ (t, i) => `Turn ${i + 1}:
796
+ User: ${t.userMessage}
797
+ Agent: ${t.agentResponse.slice(0, 1500)}`
798
+ ).join("\n\n---\n\n");
799
+ const resp = await tc.chat({
800
+ model: "gpt-4o",
801
+ messages: [
802
+ {
803
+ role: "system",
804
+ content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
805
+
806
+ 1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
807
+ 2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
808
+ 3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
809
+
810
+ Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
811
+
812
+ Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
813
+ },
814
+ {
815
+ role: "user",
816
+ content: `Persona: ${scenario.persona}
817
+ Scenario: ${scenario.thesis}
818
+
819
+ ${conversation}`
820
+ }
821
+ ],
822
+ temperature: 0.2,
823
+ maxTokens: 800
824
+ });
825
+ return parseJudgeResponse("adversarial", resp);
826
+ };
827
+ function createCustomJudge(name, systemPrompt, opts) {
828
+ return async (tc, { scenario, turns }) => {
829
+ const conversation = turns.map(
830
+ (t, i) => `Turn ${i + 1}:
831
+ User: ${t.userMessage}
832
+ Agent: ${t.agentResponse.slice(0, 2e3)}`
833
+ ).join("\n\n---\n\n");
834
+ const resp = await tc.chat({
835
+ model: opts?.model ?? "gpt-4o",
836
+ messages: [
837
+ {
838
+ role: "system",
839
+ content: systemPrompt
840
+ },
841
+ {
842
+ role: "user",
843
+ content: `Persona: ${scenario.persona} (${scenario.label})
844
+ Scenario: ${scenario.thesis}
845
+
846
+ ${conversation}`
847
+ }
848
+ ],
849
+ temperature: opts?.temperature ?? 0.1,
850
+ maxTokens: opts?.maxTokens ?? 1e3
851
+ });
852
+ return parseJudgeResponse(name, resp);
853
+ };
854
+ }
855
+ function defaultJudges(domain) {
856
+ return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
857
+ }
858
+ function parseJudgeResponse(judgeName, resp) {
859
+ const content = resp.choices?.[0]?.message?.content ?? "";
860
+ try {
861
+ let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
862
+ const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
863
+ if (arrayMatch) cleaned = arrayMatch[0];
864
+ const parsed = JSON.parse(cleaned);
865
+ return parsed.map((p) => ({
866
+ judgeName,
867
+ dimension: p.dimension,
868
+ score: Math.max(0, Math.min(10, p.score)),
869
+ reasoning: p.reasoning ?? "",
870
+ evidence: p.evidence
871
+ }));
872
+ } catch (err) {
873
+ throw new JudgeParseError(judgeName, content, { cause: err });
874
+ }
875
+ }
876
+
877
+ // src/executor.ts
878
+ async function executeScenario(tc, scenario, config) {
879
+ const startTime = Date.now();
880
+ const model = config.model ?? "gpt-4o";
881
+ const systemPrompt = [config.systemPrompt, scenario.systemPromptAppend ?? ""].filter(Boolean).join("\n\n");
882
+ const messages = [{ role: "system", content: systemPrompt }];
883
+ const turns = [];
884
+ const allCodeBlocks = [];
885
+ const allBlocks = [];
886
+ const allToolCalls = [];
887
+ const blockRe = config.blockPattern ?? /:::(\w+)\s*\n([\s\S]*?)\n\s*:::/g;
888
+ for (let i = 0; i < scenario.turns.length; i++) {
889
+ const turn = scenario.turns[i];
890
+ const turnStart = Date.now();
891
+ messages.push({ role: "user", content: turn.user });
892
+ const resp = await tc.chat({
893
+ model,
894
+ messages,
895
+ temperature: 0.4,
896
+ maxTokens: 3e3
897
+ });
898
+ const content = resp.choices?.[0]?.message?.content ?? "";
899
+ messages.push({ role: "assistant", content });
900
+ const codeRe = /```(\w+)?\n([\s\S]*?)```/g;
901
+ let codeMatch = codeRe.exec(content);
902
+ while (codeMatch !== null) {
903
+ allCodeBlocks.push({ language: codeMatch[1] ?? "text", code: codeMatch[2] ?? "" });
904
+ codeMatch = codeRe.exec(content);
905
+ }
906
+ const turnBlocks = [];
907
+ const blockReLocal = new RegExp(blockRe.source, blockRe.flags);
908
+ let blockMatch = blockReLocal.exec(content);
909
+ while (blockMatch !== null) {
910
+ const fields = {};
911
+ for (const line of (blockMatch[2] ?? "").split("\n")) {
912
+ const idx = line.indexOf(":");
913
+ if (idx > 0) fields[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
914
+ }
915
+ const blockType = blockMatch[1] ?? "";
916
+ allBlocks.push({ type: blockType, fields });
917
+ turnBlocks.push({ type: blockType, title: fields.title ?? "" });
918
+ blockMatch = blockReLocal.exec(content);
919
+ }
920
+ let hasToolCall = false;
921
+ if (config.toolCallPatterns) {
922
+ for (const pattern of config.toolCallPatterns) {
923
+ const re = new RegExp(pattern.source, pattern.flags);
924
+ let toolMatch = re.exec(content);
925
+ while (toolMatch !== null) {
926
+ allToolCalls.push(toolMatch[0]);
927
+ hasToolCall = true;
928
+ toolMatch = re.exec(content);
929
+ }
930
+ }
931
+ }
932
+ turns.push({
933
+ turnIndex: i,
934
+ userMessage: turn.user,
935
+ agentResponse: content,
936
+ durationMs: Date.now() - turnStart,
937
+ blocksExtracted: turnBlocks,
938
+ containsCode: allCodeBlocks.length > 0,
939
+ containsToolCall: hasToolCall
940
+ });
941
+ }
942
+ const artifacts = {
943
+ vaultFiles: [],
944
+ blocksExtracted: allBlocks,
945
+ codeBlocks: allCodeBlocks,
946
+ toolCalls: allToolCalls
947
+ };
948
+ const artifactResults = scenario.artifactChecks.map((check2) => {
949
+ if (config.artifactChecker) {
950
+ const custom = config.artifactChecker(check2, artifacts);
951
+ if (custom) return { check: check2, ...custom };
952
+ }
953
+ switch (check2.type) {
954
+ case "block_extracted": {
955
+ const count = allBlocks.filter((b) => b.type === check2.target).length;
956
+ return {
957
+ check: check2,
958
+ passed: count >= (check2.minCount ?? 1),
959
+ detail: `Found ${count} ${check2.target} blocks (need ${check2.minCount ?? 1})`
960
+ };
961
+ }
962
+ case "code_valid": {
963
+ const hasCode = allCodeBlocks.some(
964
+ (b) => b.language === check2.target || b.code.includes(check2.target)
965
+ );
966
+ return { check: check2, passed: hasCode, detail: hasCode ? "Code block found" : "No matching code" };
967
+ }
968
+ default:
969
+ return {
970
+ check: check2,
971
+ passed: false,
972
+ detail: `Check type "${check2.type}" requires live environment`
973
+ };
974
+ }
975
+ });
976
+ const judgeInput = { scenario, turns, artifacts };
977
+ const judgeResults = [];
978
+ let failedJudges = 0;
979
+ for (const judge of config.judges) {
980
+ for (let attempt = 0; attempt < 3; attempt++) {
981
+ try {
982
+ if (attempt > 0) {
983
+ const wait = attempt * 1e4;
984
+ console.log(` judge retry ${attempt}/2 (waiting ${wait / 1e3}s)`);
985
+ await new Promise((r) => setTimeout(r, wait));
986
+ }
987
+ const scores2 = await judge(tc, judgeInput);
988
+ judgeResults.push(scores2);
989
+ await new Promise((r) => setTimeout(r, 3e3));
990
+ break;
991
+ } catch (err) {
992
+ if (err instanceof JudgeParseError) {
993
+ failedJudges++;
994
+ break;
774
995
  }
996
+ if (attempt === 2) failedJudges++;
775
997
  }
776
998
  }
777
999
  }
@@ -799,7 +1021,7 @@ async function executeScenario(tc, scenario, config) {
799
1021
  turns,
800
1022
  artifactResults,
801
1023
  judgeScores: allScores,
802
- judgeErrors: errorScores.length,
1024
+ judgeErrors: errorScores.length + failedJudges,
803
1025
  overallScore,
804
1026
  totalDurationMs: Date.now() - startTime,
805
1027
  artifacts
@@ -1451,11 +1673,11 @@ var FileSystemFeedbackTrajectoryStore = class {
1451
1673
  }
1452
1674
  async load() {
1453
1675
  if (this.loaded) return;
1454
- const { readFile } = await import("fs/promises");
1676
+ const { readFile: readFile2 } = await import("fs/promises");
1455
1677
  const { join: join4 } = await import("path");
1456
1678
  const file = join4(this.dir, "feedback-trajectories.ndjson");
1457
1679
  try {
1458
- const raw = await readFile(file, "utf8");
1680
+ const raw = await readFile2(file, "utf8");
1459
1681
  for (const line of raw.split("\n")) {
1460
1682
  if (!line.trim()) continue;
1461
1683
  try {
@@ -1824,380 +2046,169 @@ async function preflightModels(opts) {
1824
2046
  return { succeeded: true, value: results, error: null };
1825
2047
  }
1826
2048
  var ModelsUnreachableError = class extends AgentEvalError {
1827
- constructor(message, results) {
1828
- super("config", message);
1829
- this.results = results;
1830
- this.name = "ModelsUnreachableError";
1831
- }
1832
- results;
1833
- };
1834
- function describeFailure(r) {
1835
- if (!r.listed) {
1836
- const probeNote = r.served === false ? ` (probe ${r.status}${r.detail ? `: ${r.detail}` : ""})` : "";
1837
- return `${r.model}: not in /models${probeNote}`;
1838
- }
1839
- return `${r.model}: listed but probe ${r.status}${r.detail ? ` \u2014 ${r.detail}` : ""}`;
1840
- }
1841
- async function assertModelsServed(opts) {
1842
- const outcome = await preflightModels(opts);
1843
- if (!outcome.succeeded || outcome.value === null) {
1844
- throw new ConfigError(
1845
- outcome.error ?? "assertModelsServed: preflight failed without an error message"
1846
- );
1847
- }
1848
- const dead = outcome.value.filter((r) => !r.listed || r.served === false);
1849
- if (dead.length > 0) {
1850
- throw new ModelsUnreachableError(
1851
- `assertModelsServed: ${dead.length}/${outcome.value.length} model(s) unreachable on the router \u2014 ${dead.map(describeFailure).join("; ")}`,
1852
- outcome.value
1853
- );
1854
- }
1855
- return outcome.value;
1856
- }
1857
-
1858
- // src/integrity/single-backend.ts
1859
- var SingleBackendError = class extends AgentEvalError {
1860
- constructor(message, report) {
1861
- super("backend_integrity", message);
1862
- this.report = report;
1863
- this.name = "SingleBackendError";
1864
- }
1865
- report;
1866
- };
1867
- function stripSlash2(url) {
1868
- return url.replace(/\/+$/, "");
1869
- }
1870
- function assertSingleBackend(agent, judge, opts = {}) {
1871
- const divergences = [];
1872
- if (agent.kind !== judge.kind) {
1873
- divergences.push({ field: "kind", agent: agent.kind, judge: judge.kind });
1874
- }
1875
- if (stripSlash2(agent.baseUrl) !== stripSlash2(judge.baseUrl)) {
1876
- divergences.push({ field: "baseUrl", agent: agent.baseUrl, judge: judge.baseUrl });
1877
- }
1878
- if (agent.model !== judge.model) {
1879
- divergences.push({ field: "model", agent: agent.model, judge: judge.model });
1880
- }
1881
- if (agent.provider !== judge.provider) {
1882
- divergences.push({ field: "provider", agent: agent.provider, judge: judge.provider });
1883
- }
1884
- const agentHasKey = Boolean(agent.apiKey);
1885
- const judgeHasKey = Boolean(judge.apiKey);
1886
- if (agentHasKey !== judgeHasKey) {
1887
- divergences.push({
1888
- field: "apiKeyPresence",
1889
- agent: agentHasKey ? "set" : "empty",
1890
- judge: judgeHasKey ? "set" : "empty"
1891
- });
1892
- }
1893
- const blocking = opts.strict ? divergences : divergences.filter((d) => d.field !== "model");
1894
- const ok = blocking.length === 0;
1895
- const report = { ok, divergences };
1896
- if (!ok) {
1897
- const agentLabel = opts.agentLabel ?? "agent";
1898
- const judgeLabel = opts.judgeLabel ?? "judge";
1899
- const detail = blocking.map((d) => `${d.field}: ${agentLabel}=${d.agent ?? "\u2205"} vs ${judgeLabel}=${d.judge ?? "\u2205"}`).join("; ");
1900
- throw new SingleBackendError(
1901
- `single-backend: ${agentLabel} and ${judgeLabel} backends diverge \u2014 the judge would re-route through a different backend than the agent (${detail})`,
1902
- report
1903
- );
1904
- }
1905
- return report;
1906
- }
1907
-
1908
- // src/judge-families.ts
1909
- var PROVIDER_PREFIX = {
1910
- anthropic: "anthropic",
1911
- openai: "openai",
1912
- "azure-openai": "openai",
1913
- google: "google",
1914
- "google-vertex": "google",
1915
- meta: "meta",
1916
- "meta-llama": "meta",
1917
- mistral: "mistral",
1918
- mistralai: "mistral",
1919
- deepseek: "deepseek",
1920
- xai: "xai",
1921
- qwen: "qwen",
1922
- alibaba: "qwen",
1923
- cohere: "cohere",
1924
- amazon: "amazon",
1925
- bedrock: "amazon",
1926
- moonshot: "moonshot",
1927
- moonshotai: "moonshot",
1928
- kimi: "moonshot",
1929
- "kimi-code": "moonshot",
1930
- zhipu: "zhipu",
1931
- zhipuai: "zhipu",
1932
- zai: "zhipu",
1933
- "z-ai": "zhipu",
1934
- glm: "zhipu"
1935
- };
1936
- var NAME_PATTERNS = [
1937
- [/claude/i, "anthropic"],
1938
- [/\b(gpt|davinci|babbage)\b|^o[134]\b|[-/]o[134]\b|gpt-/i, "openai"],
1939
- [/gemini|palm|gemma|bison/i, "google"],
1940
- [/llama/i, "meta"],
1941
- [/mi(s|x)tral|codestral|magistral/i, "mistral"],
1942
- [/deepseek/i, "deepseek"],
1943
- [/grok/i, "xai"],
1944
- [/qwen/i, "qwen"],
1945
- [/command-?(r|a)?/i, "cohere"],
1946
- [/\b(nova|titan)\b/i, "amazon"],
1947
- [/\bkimi\b|moonshot/i, "moonshot"],
1948
- [/\bglm\b|zhipu|\bz-?ai\b/i, "zhipu"]
1949
- ];
1950
- function judgeFamily(modelId) {
1951
- const id = modelId.trim().split("@")[0].toLowerCase();
1952
- const slash = id.indexOf("/");
1953
- if (slash > 0) {
1954
- const prefix = id.slice(0, slash);
1955
- const mapped = PROVIDER_PREFIX[prefix];
1956
- if (mapped) return mapped;
1957
- }
1958
- for (const [pattern, family] of NAME_PATTERNS) {
1959
- if (pattern.test(id)) return family;
1960
- }
1961
- return "unknown";
1962
- }
1963
- var CrossFamilyError = class extends Error {
1964
- constructor(message, families, models) {
1965
- super(message);
1966
- this.families = families;
1967
- this.models = models;
1968
- this.name = "CrossFamilyError";
1969
- }
1970
- families;
1971
- models;
1972
- };
1973
- function assertCrossFamily(models, opts = {}) {
1974
- const minFamilies = opts.minFamilies ?? 2;
1975
- const families = /* @__PURE__ */ new Set();
1976
- for (const m of models) {
1977
- const f = judgeFamily(m);
1978
- if (f === "unknown" && !opts.allowUnknown) continue;
1979
- families.add(f);
1980
- }
1981
- const list = [...families].sort();
1982
- if (list.length < minFamilies) {
1983
- throw new CrossFamilyError(
1984
- `judge ensemble spans ${list.length} provider famil${list.length === 1 ? "y" : "ies"} (${list.join(", ") || "none"}) but ${minFamilies} required \u2014 a single-family ensemble is correlated bias, not independent signal`,
1985
- list,
1986
- models
1987
- );
1988
- }
1989
- return list;
1990
- }
1991
-
1992
- // src/judges.ts
1993
- function createDomainExpertJudge(domain) {
1994
- return async (tc, { scenario, turns }) => {
1995
- const conversation = turns.map(
1996
- (t, i) => `Turn ${i + 1}:
1997
- User: ${t.userMessage}
1998
- Agent: ${t.agentResponse.slice(0, 2e3)}`
1999
- ).join("\n\n---\n\n");
2000
- const resp = await tc.chat({
2001
- model: "gpt-4o",
2002
- messages: [
2003
- {
2004
- role: "system",
2005
- content: `You are a senior ${domain} professional with 20+ years of experience. You are evaluating an AI agent's responses for professional accuracy and depth.
2006
-
2007
- Score STRICTLY. A 5 means "a junior professional could do this." An 8 means "solid mid-career work." A 10 means "I would hire this agent."
2008
-
2009
- Evaluate:
2010
- 1. **domain_accuracy** (0-10): Are the technical terms correct? Are the recommendations what you'd actually do? Would this advice cause problems if followed?
2011
- 2. **professional_depth** (0-10): Does it go beyond surface-level? Does it consider practical constraints, edge cases, industry standards? Or is it generic textbook advice?
2012
-
2013
- Respond with JSON only: [{"dimension":"domain_accuracy","score":N,"reasoning":"...","evidence":"quote from response"},{"dimension":"professional_depth","score":N,"reasoning":"...","evidence":"quote"}]`
2014
- },
2015
- {
2016
- role: "user",
2017
- content: `Persona: ${scenario.persona} (${scenario.label})
2018
- Scenario: ${scenario.thesis}
2019
-
2020
- ${conversation}`
2021
- }
2022
- ],
2023
- temperature: 0.1,
2024
- maxTokens: 800
2025
- });
2026
- return parseJudgeResponse("domain_expert", resp);
2027
- };
2028
- }
2029
- var codeExecutionJudge = async (tc, { scenario, artifacts }) => {
2030
- const codeBlocks = artifacts.codeBlocks;
2031
- if (codeBlocks.length === 0) {
2032
- return [
2033
- {
2034
- judgeName: "code_execution",
2035
- dimension: "code_execution",
2036
- score: 0,
2037
- reasoning: "No code blocks found in agent response."
2038
- }
2039
- ];
2040
- }
2041
- const codeText = codeBlocks.map(
2042
- (b, i) => `Block ${i + 1} (${b.language}):
2043
- \`\`\`${b.language}
2044
- ${b.code.slice(0, 3e3)}
2045
- \`\`\``
2046
- ).join("\n\n");
2047
- const resp = await tc.chat({
2048
- model: "gpt-4o",
2049
- messages: [
2050
- {
2051
- role: "system",
2052
- content: `You are a principal software engineer reviewing code written by an AI agent.
2053
-
2054
- Score STRICTLY:
2055
- 1. **executability** (0-10): Would this code run without errors? Check: import errors, undefined variables, missing deps, syntax errors. A 5 means "would run with minor fixes." A 10 means "copy-paste and it works."
2056
- 2. **completeness** (0-10): Does it handle the FULL task, or just the happy path? A 5 means "handles the main case." A 10 means "production-ready."
2057
- 3. **reusability** (0-10): Could this be saved as a tool and reused? A 5 means "works for this case." A 10 means "general-purpose tool."
2058
-
2059
- Respond with JSON only: [{"dimension":"executability","score":N,"reasoning":"...","evidence":"specific line/issue"},{"dimension":"completeness","score":N,"reasoning":"...","evidence":"..."},{"dimension":"reusability","score":N,"reasoning":"...","evidence":"..."}]`
2060
- },
2061
- {
2062
- role: "user",
2063
- content: `Task: ${scenario.thesis}
2064
-
2065
- ${codeText}`
2066
- }
2067
- ],
2068
- temperature: 0.1,
2069
- maxTokens: 1e3
2070
- });
2071
- return parseJudgeResponse("code_execution", resp);
2072
- };
2073
- var coherenceJudge = async (tc, { scenario, turns }) => {
2074
- if (turns.length < 2) {
2075
- return [];
2076
- }
2077
- const conversation = turns.map(
2078
- (t, i) => `Turn ${i + 1}:
2079
- User: ${t.userMessage}
2080
- Agent (${t.agentResponse.length} chars): ${t.agentResponse.slice(0, 1500)}`
2081
- ).join("\n\n---\n\n");
2082
- const resp = await tc.chat({
2083
- model: "gpt-4o",
2084
- messages: [
2085
- {
2086
- role: "system",
2087
- content: `You evaluate whether an AI agent maintains coherence across a multi-turn conversation.
2088
-
2089
- Score STRICTLY:
2090
- 1. **consistency** (0-10): Does the agent contradict itself across turns? Does it remember what it said/built earlier?
2091
- 2. **progression** (0-10): Does each turn BUILD on the previous? Or does it start fresh? A 5 means "vaguely related." A 10 means "each turn clearly advances the work."
2092
- 3. **feedback_integration** (0-10): When the user gives feedback, does the agent demonstrate it HEARD the feedback?
2093
-
2094
- Respond with JSON only: [{"dimension":"consistency","score":N,"reasoning":"..."},{"dimension":"progression","score":N,"reasoning":"..."},{"dimension":"feedback_integration","score":N,"reasoning":"..."}]`
2095
- },
2096
- {
2097
- role: "user",
2098
- content: `Scenario: ${scenario.thesis}
2099
-
2100
- ${conversation}`
2101
- }
2102
- ],
2103
- temperature: 0.1,
2104
- maxTokens: 800
2105
- });
2106
- return parseJudgeResponse("coherence", resp);
2107
- };
2108
- var adversarialJudge = async (tc, { scenario, turns }) => {
2109
- const conversation = turns.map(
2110
- (t, i) => `Turn ${i + 1}:
2111
- User: ${t.userMessage}
2112
- Agent: ${t.agentResponse.slice(0, 1500)}`
2113
- ).join("\n\n---\n\n");
2114
- const resp = await tc.chat({
2115
- model: "gpt-4o",
2116
- messages: [
2117
- {
2118
- role: "system",
2119
- content: `You are a red-team evaluator. Find the WEAKEST point in the AI agent's responses. Be adversarial.
2120
-
2121
- 1. **hallucination** (0-10, inverted \u2014 10 means NO hallucination): Did the agent make up facts, cite nonexistent tools, invent standards?
2122
- 2. **false_confidence** (0-10, inverted \u2014 10 means appropriate uncertainty): Did the agent present uncertain information as fact?
2123
- 3. **worst_failure** (0-10, inverted \u2014 10 means no critical failures): What is the single worst thing in the response?
2124
-
2125
- Be harsh. If everything is genuinely good, say so \u2014 but look hard first.
2126
-
2127
- Respond with JSON only: [{"dimension":"hallucination","score":N,"reasoning":"...","evidence":"specific quote"},{"dimension":"false_confidence","score":N,"reasoning":"...","evidence":"..."},{"dimension":"worst_failure","score":N,"reasoning":"...","evidence":"..."}]`
2128
- },
2129
- {
2130
- role: "user",
2131
- content: `Persona: ${scenario.persona}
2132
- Scenario: ${scenario.thesis}
2133
-
2134
- ${conversation}`
2135
- }
2136
- ],
2137
- temperature: 0.2,
2138
- maxTokens: 800
2139
- });
2140
- return parseJudgeResponse("adversarial", resp);
2049
+ constructor(message, results) {
2050
+ super("config", message);
2051
+ this.results = results;
2052
+ this.name = "ModelsUnreachableError";
2053
+ }
2054
+ results;
2141
2055
  };
2142
- function createCustomJudge(name, systemPrompt, opts) {
2143
- return async (tc, { scenario, turns }) => {
2144
- const conversation = turns.map(
2145
- (t, i) => `Turn ${i + 1}:
2146
- User: ${t.userMessage}
2147
- Agent: ${t.agentResponse.slice(0, 2e3)}`
2148
- ).join("\n\n---\n\n");
2149
- const resp = await tc.chat({
2150
- model: opts?.model ?? "gpt-4o",
2151
- messages: [
2152
- {
2153
- role: "system",
2154
- content: systemPrompt
2155
- },
2156
- {
2157
- role: "user",
2158
- content: `Persona: ${scenario.persona} (${scenario.label})
2159
- Scenario: ${scenario.thesis}
2056
+ function describeFailure(r) {
2057
+ if (!r.listed) {
2058
+ const probeNote = r.served === false ? ` (probe ${r.status}${r.detail ? `: ${r.detail}` : ""})` : "";
2059
+ return `${r.model}: not in /models${probeNote}`;
2060
+ }
2061
+ return `${r.model}: listed but probe ${r.status}${r.detail ? ` \u2014 ${r.detail}` : ""}`;
2062
+ }
2063
+ async function assertModelsServed(opts) {
2064
+ const outcome = await preflightModels(opts);
2065
+ if (!outcome.succeeded || outcome.value === null) {
2066
+ throw new ConfigError(
2067
+ outcome.error ?? "assertModelsServed: preflight failed without an error message"
2068
+ );
2069
+ }
2070
+ const dead = outcome.value.filter((r) => !r.listed || r.served === false);
2071
+ if (dead.length > 0) {
2072
+ throw new ModelsUnreachableError(
2073
+ `assertModelsServed: ${dead.length}/${outcome.value.length} model(s) unreachable on the router \u2014 ${dead.map(describeFailure).join("; ")}`,
2074
+ outcome.value
2075
+ );
2076
+ }
2077
+ return outcome.value;
2078
+ }
2160
2079
 
2161
- ${conversation}`
2162
- }
2163
- ],
2164
- temperature: opts?.temperature ?? 0.1,
2165
- maxTokens: opts?.maxTokens ?? 1e3
2080
+ // src/integrity/single-backend.ts
2081
+ var SingleBackendError = class extends AgentEvalError {
2082
+ constructor(message, report) {
2083
+ super("backend_integrity", message);
2084
+ this.report = report;
2085
+ this.name = "SingleBackendError";
2086
+ }
2087
+ report;
2088
+ };
2089
+ function stripSlash2(url) {
2090
+ return url.replace(/\/+$/, "");
2091
+ }
2092
+ function assertSingleBackend(agent, judge, opts = {}) {
2093
+ const divergences = [];
2094
+ if (agent.kind !== judge.kind) {
2095
+ divergences.push({ field: "kind", agent: agent.kind, judge: judge.kind });
2096
+ }
2097
+ if (stripSlash2(agent.baseUrl) !== stripSlash2(judge.baseUrl)) {
2098
+ divergences.push({ field: "baseUrl", agent: agent.baseUrl, judge: judge.baseUrl });
2099
+ }
2100
+ if (agent.model !== judge.model) {
2101
+ divergences.push({ field: "model", agent: agent.model, judge: judge.model });
2102
+ }
2103
+ if (agent.provider !== judge.provider) {
2104
+ divergences.push({ field: "provider", agent: agent.provider, judge: judge.provider });
2105
+ }
2106
+ const agentHasKey = Boolean(agent.apiKey);
2107
+ const judgeHasKey = Boolean(judge.apiKey);
2108
+ if (agentHasKey !== judgeHasKey) {
2109
+ divergences.push({
2110
+ field: "apiKeyPresence",
2111
+ agent: agentHasKey ? "set" : "empty",
2112
+ judge: judgeHasKey ? "set" : "empty"
2166
2113
  });
2167
- return parseJudgeResponse(name, resp);
2168
- };
2114
+ }
2115
+ const blocking = opts.strict ? divergences : divergences.filter((d) => d.field !== "model");
2116
+ const ok = blocking.length === 0;
2117
+ const report = { ok, divergences };
2118
+ if (!ok) {
2119
+ const agentLabel = opts.agentLabel ?? "agent";
2120
+ const judgeLabel = opts.judgeLabel ?? "judge";
2121
+ const detail = blocking.map((d) => `${d.field}: ${agentLabel}=${d.agent ?? "\u2205"} vs ${judgeLabel}=${d.judge ?? "\u2205"}`).join("; ");
2122
+ throw new SingleBackendError(
2123
+ `single-backend: ${agentLabel} and ${judgeLabel} backends diverge \u2014 the judge would re-route through a different backend than the agent (${detail})`,
2124
+ report
2125
+ );
2126
+ }
2127
+ return report;
2169
2128
  }
2170
- function defaultJudges(domain) {
2171
- return [createDomainExpertJudge(domain), codeExecutionJudge, coherenceJudge, adversarialJudge];
2129
+
2130
+ // src/judge-families.ts
2131
+ var PROVIDER_PREFIX = {
2132
+ anthropic: "anthropic",
2133
+ openai: "openai",
2134
+ "azure-openai": "openai",
2135
+ google: "google",
2136
+ "google-vertex": "google",
2137
+ meta: "meta",
2138
+ "meta-llama": "meta",
2139
+ mistral: "mistral",
2140
+ mistralai: "mistral",
2141
+ deepseek: "deepseek",
2142
+ xai: "xai",
2143
+ qwen: "qwen",
2144
+ alibaba: "qwen",
2145
+ cohere: "cohere",
2146
+ amazon: "amazon",
2147
+ bedrock: "amazon",
2148
+ moonshot: "moonshot",
2149
+ moonshotai: "moonshot",
2150
+ kimi: "moonshot",
2151
+ "kimi-code": "moonshot",
2152
+ zhipu: "zhipu",
2153
+ zhipuai: "zhipu",
2154
+ zai: "zhipu",
2155
+ "z-ai": "zhipu",
2156
+ glm: "zhipu"
2157
+ };
2158
+ var NAME_PATTERNS = [
2159
+ [/claude/i, "anthropic"],
2160
+ [/\b(gpt|davinci|babbage)\b|^o[134]\b|[-/]o[134]\b|gpt-/i, "openai"],
2161
+ [/gemini|palm|gemma|bison/i, "google"],
2162
+ [/llama/i, "meta"],
2163
+ [/mi(s|x)tral|codestral|magistral/i, "mistral"],
2164
+ [/deepseek/i, "deepseek"],
2165
+ [/grok/i, "xai"],
2166
+ [/qwen/i, "qwen"],
2167
+ [/command-?(r|a)?/i, "cohere"],
2168
+ [/\b(nova|titan)\b/i, "amazon"],
2169
+ [/\bkimi\b|moonshot/i, "moonshot"],
2170
+ [/\bglm\b|zhipu|\bz-?ai\b/i, "zhipu"]
2171
+ ];
2172
+ function judgeFamily(modelId) {
2173
+ const id = modelId.trim().split("@")[0].toLowerCase();
2174
+ const slash = id.indexOf("/");
2175
+ if (slash > 0) {
2176
+ const prefix = id.slice(0, slash);
2177
+ const mapped = PROVIDER_PREFIX[prefix];
2178
+ if (mapped) return mapped;
2179
+ }
2180
+ for (const [pattern, family] of NAME_PATTERNS) {
2181
+ if (pattern.test(id)) return family;
2182
+ }
2183
+ return "unknown";
2172
2184
  }
2173
- function parseJudgeResponse(judgeName, resp) {
2174
- try {
2175
- const content = resp.choices?.[0]?.message?.content ?? "";
2176
- let cleaned = content.replace(/```json\n?|\n?```/g, "").trim();
2177
- const arrayMatch = cleaned.match(/\[[\s\S]*\]/);
2178
- if (arrayMatch) cleaned = arrayMatch[0];
2179
- const parsed = JSON.parse(cleaned);
2180
- return parsed.map((p) => ({
2181
- judgeName,
2182
- dimension: p.dimension,
2183
- score: Math.max(0, Math.min(10, p.score)),
2184
- reasoning: p.reasoning ?? "",
2185
- evidence: p.evidence
2186
- }));
2187
- } catch (err) {
2188
- const content = resp.choices?.[0]?.message?.content ?? "";
2189
- console.log(
2190
- ` [parse_error] ${judgeName}: ${err.message?.slice(0, 50)} | response: ${content.slice(0, 100)}`
2185
+ var CrossFamilyError = class extends Error {
2186
+ constructor(message, families, models) {
2187
+ super(message);
2188
+ this.families = families;
2189
+ this.models = models;
2190
+ this.name = "CrossFamilyError";
2191
+ }
2192
+ families;
2193
+ models;
2194
+ };
2195
+ function assertCrossFamily(models, opts = {}) {
2196
+ const minFamilies = opts.minFamilies ?? 2;
2197
+ const families = /* @__PURE__ */ new Set();
2198
+ for (const m of models) {
2199
+ const f = judgeFamily(m);
2200
+ if (f === "unknown" && !opts.allowUnknown) continue;
2201
+ families.add(f);
2202
+ }
2203
+ const list = [...families].sort();
2204
+ if (list.length < minFamilies) {
2205
+ throw new CrossFamilyError(
2206
+ `judge ensemble spans ${list.length} provider famil${list.length === 1 ? "y" : "ies"} (${list.join(", ") || "none"}) but ${minFamilies} required \u2014 a single-family ensemble is correlated bias, not independent signal`,
2207
+ list,
2208
+ models
2191
2209
  );
2192
- return [
2193
- {
2194
- judgeName,
2195
- dimension: "parse_error",
2196
- score: 0,
2197
- reasoning: `Parse failed: ${err.message?.slice(0, 100)}. Raw: ${content.slice(0, 200)}`
2198
- }
2199
- ];
2200
2210
  }
2211
+ return list;
2201
2212
  }
2202
2213
 
2203
2214
  // src/live-proof.ts
@@ -2994,17 +3005,181 @@ var DualAgentBench = class {
2994
3005
  finalScore: lastScore
2995
3006
  });
2996
3007
  }
2997
- const convergedResults = results.filter((r) => r.converged);
2998
- const convergenceRate = results.length ? convergedResults.length / results.length : 0;
2999
- const avgRoundsToConverge = convergedResults.length ? convergedResults.reduce((acc, r) => acc + (r.roundsToConverge ?? 0), 0) / convergedResults.length : null;
3000
- const avgFinalScore = results.length ? results.reduce((acc, r) => acc + r.finalScore, 0) / results.length : 0;
3001
- return {
3002
- scenarios: results,
3003
- aggregate: { convergenceRate, avgRoundsToConverge, avgFinalScore },
3004
- config: { maxRounds, convergenceThreshold: threshold }
3005
- };
3008
+ const convergedResults = results.filter((r) => r.converged);
3009
+ const convergenceRate = results.length ? convergedResults.length / results.length : 0;
3010
+ const avgRoundsToConverge = convergedResults.length ? convergedResults.reduce((acc, r) => acc + (r.roundsToConverge ?? 0), 0) / convergedResults.length : null;
3011
+ const avgFinalScore = results.length ? results.reduce((acc, r) => acc + r.finalScore, 0) / results.length : 0;
3012
+ return {
3013
+ scenarios: results,
3014
+ aggregate: { convergenceRate, avgRoundsToConverge, avgFinalScore },
3015
+ config: { maxRounds, convergenceThreshold: threshold }
3016
+ };
3017
+ }
3018
+ };
3019
+
3020
+ // src/eval-tools.ts
3021
+ import { readFile } from "fs/promises";
3022
+ function toOpenAiTool(def) {
3023
+ return {
3024
+ type: "function",
3025
+ function: { name: def.name, description: def.description, parameters: def.parameters }
3026
+ };
3027
+ }
3028
+ function makeEvalTools(cfg) {
3029
+ const tools = [];
3030
+ if (cfg.judges) tools.push(runJudgesTool(cfg.judges));
3031
+ if (cfg.completion) tools.push(verifyCompletionTool(cfg.completion.checkCorrectness));
3032
+ if (cfg.analyze) tools.push(analyzeRunsTool(cfg.analyze));
3033
+ return tools;
3034
+ }
3035
+ function runJudgesTool(judges) {
3036
+ if (judges.length === 0) {
3037
+ throw new Error("makeEvalTools: cfg.judges is empty \u2014 supply judges or omit the section");
3038
+ }
3039
+ const names = judges.map((j) => j.name);
3040
+ return {
3041
+ name: "run_judges",
3042
+ description: `Score an artifact with the configured judges (${names.join(", ")}). Pass \`judge\` to run one judge by name; omit it to run all. Returns per-judge scores keyed by judge name.`,
3043
+ parameters: {
3044
+ type: "object",
3045
+ properties: {
3046
+ judge: {
3047
+ type: "string",
3048
+ enum: names,
3049
+ description: "Run only this judge. Omit to run every configured judge."
3050
+ },
3051
+ artifact: {
3052
+ description: "The artifact to score \u2014 any JSON value the judges understand."
3053
+ },
3054
+ scenario: {
3055
+ description: "Optional scenario context forwarded to the judges."
3056
+ }
3057
+ },
3058
+ required: ["artifact"],
3059
+ additionalProperties: false
3060
+ },
3061
+ handler: async (args, ctx) => {
3062
+ const a = requireObjectArgs("run_judges", args);
3063
+ if (!("artifact" in a)) {
3064
+ throw new Error("run_judges: args.artifact is required");
3065
+ }
3066
+ let selected = judges;
3067
+ if (a.judge !== void 0) {
3068
+ if (typeof a.judge !== "string") throw new Error("run_judges: args.judge must be a string");
3069
+ selected = judges.filter((j) => j.name === a.judge);
3070
+ if (selected.length === 0) {
3071
+ throw new Error(
3072
+ `run_judges: unknown judge '${a.judge}' \u2014 configured: ${names.join(", ")}`
3073
+ );
3074
+ }
3075
+ }
3076
+ const signal = ctx?.signal ?? new AbortController().signal;
3077
+ const scenario = a.scenario;
3078
+ const scores2 = {};
3079
+ for (const judge of selected) {
3080
+ if (scenario !== void 0 && judge.appliesTo && !judge.appliesTo(scenario)) continue;
3081
+ scores2[judge.name] = await judge.score({ artifact: a.artifact, scenario, signal });
3082
+ }
3083
+ return { scores: scores2 };
3084
+ }
3085
+ };
3086
+ }
3087
+ function verifyCompletionTool(checkCorrectness) {
3088
+ return {
3089
+ name: "verify_completion",
3090
+ description: "Verify produced state against a gold task spec: each gold requirement is matched to at most one produced artifact/proposal/tool-call, correctness-checked by the host, and reduced to a CompletionVerdict (completionRate, fullyComplete, per-requirement checks).",
3091
+ parameters: {
3092
+ type: "object",
3093
+ properties: {
3094
+ gold: {
3095
+ type: "object",
3096
+ description: "TaskGold \u2014 taskId + requirements the produced state must satisfy."
3097
+ },
3098
+ state: {
3099
+ type: "object",
3100
+ description: "ProducedState \u2014 artifacts, proposals, and tool calls the agent produced."
3101
+ }
3102
+ },
3103
+ required: ["gold", "state"],
3104
+ additionalProperties: false
3105
+ },
3106
+ handler: async (args) => {
3107
+ const a = requireObjectArgs("verify_completion", args);
3108
+ if (typeof a.gold !== "object" || a.gold === null) {
3109
+ throw new Error("verify_completion: args.gold (TaskGold) is required");
3110
+ }
3111
+ if (typeof a.state !== "object" || a.state === null) {
3112
+ throw new Error("verify_completion: args.state (ProducedState) is required");
3113
+ }
3114
+ return verifyCompletion(a.gold, a.state, checkCorrectness);
3115
+ }
3116
+ };
3117
+ }
3118
+ function analyzeRunsTool(analyzeOpts) {
3119
+ return {
3120
+ name: "analyze_runs",
3121
+ description: "Run analyzeRuns over a set of RunRecords and return the InsightReport (composite distribution, per-dimension stats, lift, failure clusters, recommendations). Pass the records inline via `runs`, or `path` to a .json (array) or .jsonl (one record per line) file on the host.",
3122
+ parameters: {
3123
+ type: "object",
3124
+ properties: {
3125
+ runs: {
3126
+ type: "array",
3127
+ items: { type: "object" },
3128
+ description: "RunRecord[] inline."
3129
+ },
3130
+ path: {
3131
+ type: "string",
3132
+ description: "Host path to a JSON array or JSONL file of RunRecords."
3133
+ }
3134
+ },
3135
+ additionalProperties: false
3136
+ },
3137
+ handler: async (args) => {
3138
+ const a = requireObjectArgs("analyze_runs", args);
3139
+ const hasRuns = Array.isArray(a.runs);
3140
+ const hasPath = typeof a.path === "string" && a.path.length > 0;
3141
+ if (hasRuns === hasPath) {
3142
+ throw new Error("analyze_runs: pass exactly one of args.runs (array) or args.path (string)");
3143
+ }
3144
+ const runs = hasRuns ? a.runs : await loadRunRecords(a.path);
3145
+ if (runs.length === 0) {
3146
+ throw new Error("analyze_runs: no runs to analyze");
3147
+ }
3148
+ return analyzeRuns({ ...analyzeOpts, runs });
3149
+ }
3150
+ };
3151
+ }
3152
+ async function loadRunRecords(path) {
3153
+ const text = await readFile(path, "utf8");
3154
+ const trimmed = text.trim();
3155
+ if (trimmed.length === 0) {
3156
+ throw new Error(`analyze_runs: file '${path}' is empty`);
3157
+ }
3158
+ if (trimmed.startsWith("[")) {
3159
+ const parsed = JSON.parse(trimmed);
3160
+ if (!Array.isArray(parsed)) {
3161
+ throw new Error(`analyze_runs: file '${path}' did not parse to an array`);
3162
+ }
3163
+ return parsed;
3006
3164
  }
3007
- };
3165
+ return trimmed.split("\n").flatMap((line, i) => {
3166
+ const l = line.trim();
3167
+ if (l.length === 0) return [];
3168
+ try {
3169
+ return [JSON.parse(l)];
3170
+ } catch (err) {
3171
+ throw new Error(
3172
+ `analyze_runs: file '${path}' line ${i + 1} is not valid JSON: ${err instanceof Error ? err.message : String(err)}`
3173
+ );
3174
+ }
3175
+ });
3176
+ }
3177
+ function requireObjectArgs(tool, args) {
3178
+ if (typeof args !== "object" || args === null || Array.isArray(args)) {
3179
+ throw new Error(`${tool}: args must be an object`);
3180
+ }
3181
+ return args;
3182
+ }
3008
3183
 
3009
3184
  // src/harness-optimizer.ts
3010
3185
  var DEFAULT_HARNESS_OBJECTIVES = [
@@ -3127,15 +3302,22 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3127
3302
  for (const d of dimensionKeys) dimAcc[d] = [];
3128
3303
  let rationale = "";
3129
3304
  let costUsd = 0;
3305
+ const seenCount = /* @__PURE__ */ new Map();
3306
+ const keyFor = (model) => {
3307
+ const n = (seenCount.get(model) ?? 0) + 1;
3308
+ seenCount.set(model, n);
3309
+ return n === 1 ? model : `${model}#${n}`;
3310
+ };
3130
3311
  for (const v of verdicts) {
3131
3312
  costUsd += v.costUsd ?? 0;
3313
+ const key = keyFor(v.model);
3132
3314
  if (!v.perDimension) {
3133
- failedJudges.push(v.model);
3315
+ failedJudges.push(key);
3134
3316
  continue;
3135
3317
  }
3136
3318
  const dims = {};
3137
3319
  for (const d of dimensionKeys) dims[d] = clamp01(Number(v.perDimension[d]));
3138
- perJudge[v.model] = dims;
3320
+ perJudge[key] = dims;
3139
3321
  for (const d of dimensionKeys) dimAcc[d].push(dims[d]);
3140
3322
  if (!rationale && typeof v.rationale === "string" && v.rationale) rationale = v.rationale;
3141
3323
  }
@@ -3170,7 +3352,127 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3170
3352
  maxDisagreement,
3171
3353
  failedJudges,
3172
3354
  costUsd,
3173
- rationale: rationale || "llm-judge"
3355
+ rationale: rationale || "llm-judge",
3356
+ verdicts: [...verdicts]
3357
+ };
3358
+ }
3359
+
3360
+ // src/judge-retry.ts
3361
+ var DEFAULT_MAX_ATTEMPTS = 3;
3362
+ var DEFAULT_TIMEOUT_MS = 3e5;
3363
+ function sleep(ms) {
3364
+ return new Promise((resolve) => setTimeout(resolve, ms));
3365
+ }
3366
+ async function withJudgeRetry(judgeFn, policy = {}) {
3367
+ const maxAttempts = policy.maxAttempts ?? DEFAULT_MAX_ATTEMPTS;
3368
+ const timeoutMs = policy.timeoutMs ?? DEFAULT_TIMEOUT_MS;
3369
+ const backoff = policy.backoffMs ?? backoffMs;
3370
+ const isRetryable = policy.isRetryable ?? isTransientLlmError;
3371
+ const models = policy.models && policy.models.length > 0 ? policy.models : [void 0];
3372
+ let totalAttempts = 0;
3373
+ const attemptErrors = [];
3374
+ let lastError;
3375
+ for (const model of models) {
3376
+ for (let attempt = 0; attempt < maxAttempts; attempt++) {
3377
+ totalAttempts += 1;
3378
+ const controller = new AbortController();
3379
+ const timer = setTimeout(() => controller.abort(new Error("TimeoutError")), timeoutMs);
3380
+ try {
3381
+ const value = await judgeFn(model, controller.signal);
3382
+ clearTimeout(timer);
3383
+ return {
3384
+ value,
3385
+ succeeded: true,
3386
+ attempts: totalAttempts,
3387
+ modelUsed: model,
3388
+ attemptErrors
3389
+ };
3390
+ } catch (err) {
3391
+ clearTimeout(timer);
3392
+ const errObj = err instanceof Error ? err : new Error(String(err));
3393
+ lastError = errObj;
3394
+ attemptErrors.push({
3395
+ attempt: totalAttempts,
3396
+ model: model ?? "(default)",
3397
+ error: errObj.message
3398
+ });
3399
+ if (!isRetryable(errObj)) {
3400
+ return {
3401
+ value: null,
3402
+ succeeded: false,
3403
+ attempts: totalAttempts,
3404
+ error: errObj,
3405
+ attemptErrors
3406
+ };
3407
+ }
3408
+ if (attempt < maxAttempts - 1) {
3409
+ await sleep(backoff(attempt));
3410
+ }
3411
+ }
3412
+ }
3413
+ }
3414
+ return {
3415
+ value: null,
3416
+ succeeded: false,
3417
+ attempts: totalAttempts,
3418
+ error: lastError,
3419
+ attemptErrors
3420
+ };
3421
+ }
3422
+
3423
+ // src/judge-panel.ts
3424
+ function ensembleJudge(opts) {
3425
+ if (opts.models.length === 0) {
3426
+ throw new Error(`ensembleJudge '${opts.name}': models is empty \u2014 nothing to score with`);
3427
+ }
3428
+ if (opts.dimensions.length === 0) {
3429
+ throw new Error(`ensembleJudge '${opts.name}': dimensions is empty \u2014 nothing to score`);
3430
+ }
3431
+ if (opts.crossFamily !== false) {
3432
+ assertCrossFamily(opts.models);
3433
+ }
3434
+ const scoreOne = async (model, input) => {
3435
+ if (opts.retry) {
3436
+ const outcome = await withJudgeRetry((m) => opts.scoreWith(m, input), {
3437
+ ...opts.retry,
3438
+ models: [model]
3439
+ });
3440
+ if (!outcome.succeeded || outcome.value === null) {
3441
+ return {
3442
+ model,
3443
+ perDimension: null,
3444
+ rationale: outcome.error?.message ?? "judge failed after retries"
3445
+ };
3446
+ }
3447
+ return outcome.value;
3448
+ }
3449
+ try {
3450
+ return await opts.scoreWith(model, input);
3451
+ } catch (err) {
3452
+ return {
3453
+ model,
3454
+ perDimension: null,
3455
+ rationale: err instanceof Error ? err.message : String(err)
3456
+ };
3457
+ }
3458
+ };
3459
+ return {
3460
+ name: opts.name,
3461
+ dimensions: opts.dimensions.map((d) => ({ key: d, description: d })),
3462
+ async score({ artifact, scenario }) {
3463
+ const input = { artifact, scenario };
3464
+ const verdicts = await Promise.all(opts.models.map((model) => scoreOne(model, input)));
3465
+ const agg = aggregateJudgeVerdicts(verdicts, opts.dimensions, opts.weights);
3466
+ const score = {
3467
+ dimensions: agg.perDimension,
3468
+ composite: agg.composite,
3469
+ notes: agg.rationale,
3470
+ maxDisagreement: agg.maxDisagreement,
3471
+ perJudge: agg.perJudge
3472
+ };
3473
+ if (agg.failedJudges.length > 0) score.failedJudges = agg.failedJudges;
3474
+ return score;
3475
+ }
3174
3476
  };
3175
3477
  }
3176
3478
 
@@ -3505,7 +3807,7 @@ function createAxService(provider, apiKey, model) {
3505
3807
  });
3506
3808
  }
3507
3809
  function seededShuffle(items, seed) {
3508
- const rng = mulberry32(hashString(seed));
3810
+ const rng = mulberry322(hashString(seed));
3509
3811
  const out = [...items];
3510
3812
  for (let i = out.length - 1; i > 0; i--) {
3511
3813
  const j = Math.floor(rng() * (i + 1));
@@ -3521,7 +3823,7 @@ function hashString(value) {
3521
3823
  }
3522
3824
  return h >>> 0;
3523
3825
  }
3524
- function mulberry32(seed) {
3826
+ function mulberry322(seed) {
3525
3827
  let a = seed >>> 0;
3526
3828
  return () => {
3527
3829
  a = a + 1831565813 | 0;
@@ -4856,42 +5158,6 @@ function formatScorecardDiff(diff) {
4856
5158
  return lines.join("\n");
4857
5159
  }
4858
5160
 
4859
- // src/series-convergence.ts
4860
- function analyzeSeries(values, options = {}) {
4861
- const window = options.window ?? 5;
4862
- const stableCv = options.stableCv ?? 0.05;
4863
- const driftRun = options.driftRun ?? 3;
4864
- if (values.length < Math.max(2, Math.min(window, 3))) {
4865
- return { state: "insufficient-data", windowMean: 0, windowCv: 0, tailRun: 0, stable: false };
4866
- }
4867
- const tail = values.slice(-window);
4868
- const mean5 = tail.reduce((a, b) => a + b, 0) / tail.length;
4869
- const variance = tail.reduce((acc, v) => acc + (v - mean5) ** 2, 0) / tail.length;
4870
- const stdDev = Math.sqrt(variance);
4871
- const refMean = Math.abs(mean5) > 1e-9 ? Math.abs(mean5) : 1;
4872
- const cv = stdDev / refMean;
4873
- const stable = tail.length >= window && cv <= stableCv;
4874
- let tailRun = 0;
4875
- let direction = 0;
4876
- for (let i = values.length - 1; i > 0; i--) {
4877
- const delta = values[i] - values[i - 1];
4878
- if (delta === 0) break;
4879
- const dir = delta > 0 ? 1 : -1;
4880
- if (direction === 0) direction = dir;
4881
- if (dir !== direction) break;
4882
- tailRun += dir;
4883
- }
4884
- let state;
4885
- if (stable) {
4886
- state = "stabilized";
4887
- } else if (Math.abs(tailRun) >= driftRun) {
4888
- state = tailRun > 0 ? "drifting-up" : "drifting-down";
4889
- } else {
4890
- state = "noisy";
4891
- }
4892
- return { state, windowMean: mean5, windowCv: cv, tailRun, stable };
4893
- }
4894
-
4895
5161
  // src/slo.ts
4896
5162
  function checkSlos(metrics, slos) {
4897
5163
  const results = slos.map((slo) => check(slo, metrics[slo.metric]));
@@ -4978,70 +5244,445 @@ function scoreContinuity(pair, checks, options = {}) {
4978
5244
  if (checks.length === 0) {
4979
5245
  throw new Error("scoreContinuity: at least 1 check required");
4980
5246
  }
4981
- const passThreshold = options.passThreshold ?? 0.8;
4982
- const results = checks.map((c) => {
4983
- const raw = c.score(pair);
4984
- const clamped = Number.isFinite(raw) ? Math.max(0, Math.min(1, raw)) : 0;
4985
- return { id: c.id, description: c.description, score: clamped, pass: clamped >= passThreshold };
4986
- });
4987
- const overallScore = results.reduce((a, r) => a + r.score, 0) / results.length;
4988
- return { results, overallScore, pass: results.every((r) => r.pass) };
5247
+ const passThreshold = options.passThreshold ?? 0.8;
5248
+ const results = checks.map((c) => {
5249
+ const raw = c.score(pair);
5250
+ const clamped = Number.isFinite(raw) ? Math.max(0, Math.min(1, raw)) : 0;
5251
+ return { id: c.id, description: c.description, score: clamped, pass: clamped >= passThreshold };
5252
+ });
5253
+ const overallScore = results.reduce((a, r) => a + r.score, 0) / results.length;
5254
+ return { results, overallScore, pass: results.every((r) => r.pass) };
5255
+ }
5256
+ function keyPreserved(key) {
5257
+ return {
5258
+ id: `preserved(${key})`,
5259
+ description: `"${key}" unchanged from before to after`,
5260
+ score: ({ before, after }) => before[key] !== void 0 && before[key] === after[key] ? 1 : 0
5261
+ };
5262
+ }
5263
+ function collectionPreserved(key, minRatio = 1) {
5264
+ return {
5265
+ id: `collection-preserved(${key})`,
5266
+ description: `"${key}" length \u2265 ${minRatio} \xD7 prior length`,
5267
+ score: ({ before, after }) => {
5268
+ const b = before[key];
5269
+ const a = after[key];
5270
+ if (!Array.isArray(b) || !Array.isArray(a)) return 0;
5271
+ if (b.length === 0) return a.length === 0 ? 1 : 1;
5272
+ return Math.min(1, a.length / (b.length * minRatio));
5273
+ }
5274
+ };
5275
+ }
5276
+ function statusAdvanced(key, progression) {
5277
+ return {
5278
+ id: `status-advanced(${key})`,
5279
+ description: `"${key}" progressed along ${progression.join("\u2192")}`,
5280
+ score: ({ before, after }) => {
5281
+ const bi = progression.indexOf(String(before[key]));
5282
+ const ai2 = progression.indexOf(String(after[key]));
5283
+ if (bi === -1 || ai2 === -1) return 0;
5284
+ return ai2 >= bi ? 1 : 0;
5285
+ }
5286
+ };
5287
+ }
5288
+
5289
+ // src/ui-finding.ts
5290
+ var UI_LENSES = [
5291
+ "consistency",
5292
+ "hierarchy",
5293
+ "layout",
5294
+ "ux-flow",
5295
+ "duplication",
5296
+ "accessibility",
5297
+ "responsive",
5298
+ "states",
5299
+ "content",
5300
+ "interaction",
5301
+ "performance-perceived",
5302
+ "other"
5303
+ ];
5304
+ var UI_FINDING_SEVERITIES = [
5305
+ "critical",
5306
+ "high",
5307
+ "med",
5308
+ "low"
5309
+ ];
5310
+
5311
+ // src/trace-contracts.ts
5312
+ function isSerializedRegex(v) {
5313
+ return typeof v === "object" && v !== null && typeof v.$regex === "string" && typeof v.flags === "string";
5314
+ }
5315
+ function matchText(actual, matcher) {
5316
+ if (typeof actual !== "string") return false;
5317
+ if (typeof matcher === "string") return actual === matcher;
5318
+ if (matcher instanceof RegExp) return matcher.test(actual);
5319
+ return new RegExp(matcher.$regex, matcher.flags).test(actual);
5320
+ }
5321
+ function resolveToolName(span) {
5322
+ if (typeof span.toolName === "string") return span.toolName;
5323
+ const fromAttr = span.attributes?.["tool.name"] ?? span.attributes?.["toolName"];
5324
+ if (typeof fromAttr === "string") return fromAttr;
5325
+ if (span.kind === "tool" && typeof span.name === "string") return span.name;
5326
+ return void 0;
5327
+ }
5328
+ function assertPredicate(value, where) {
5329
+ if (value === null || typeof value !== "object") {
5330
+ throw new ValidationError(`${where}: predicate must be an object, got ${typeof value}`);
5331
+ }
5332
+ const p = value;
5333
+ for (const field of ["name", "tool"]) {
5334
+ const m = p[field];
5335
+ if (m !== void 0 && typeof m !== "string" && !(m instanceof RegExp) && !isSerializedRegex(m)) {
5336
+ throw new ValidationError(`${where}: "${field}" must be string | RegExp | SerializedRegex`);
5337
+ }
5338
+ }
5339
+ if (p.attr !== void 0 && (p.attr === null || typeof p.attr !== "object" || Array.isArray(p.attr))) {
5340
+ throw new ValidationError(`${where}: "attr" must be a plain object`);
5341
+ }
5342
+ if (p.requiresCustom && typeof p.custom !== "function") {
5343
+ throw new ValidationError(
5344
+ `${where}: rule was built with a custom predicate function, which does not survive JSON serialization \u2014 re-attach \`custom\` after deserializing or drop the rule`
5345
+ );
5346
+ }
5347
+ const hasAttr = p.attr !== void 0 && Object.keys(p.attr).length > 0;
5348
+ if (p.name === void 0 && p.tool === void 0 && !hasAttr && typeof p.custom !== "function") {
5349
+ throw new ValidationError(
5350
+ `${where}: empty predicate would match every span \u2014 specify name, tool, attr, or custom`
5351
+ );
5352
+ }
5353
+ }
5354
+ function matchSpan(span, predicate) {
5355
+ assertPredicate(predicate, "matchSpan");
5356
+ if (predicate.name !== void 0 && !matchText(span.name, predicate.name)) return false;
5357
+ if (predicate.tool !== void 0 && !matchText(resolveToolName(span), predicate.tool))
5358
+ return false;
5359
+ if (predicate.attr !== void 0) {
5360
+ for (const [key, expected] of Object.entries(predicate.attr)) {
5361
+ const actual = span.attributes?.[key];
5362
+ if (expected instanceof RegExp || isSerializedRegex(expected)) {
5363
+ if (!matchText(typeof actual === "string" ? actual : void 0, expected)) {
5364
+ return false;
5365
+ }
5366
+ } else if (actual !== expected) {
5367
+ return false;
5368
+ }
5369
+ }
5370
+ }
5371
+ if (predicate.custom !== void 0 && !predicate.custom(span)) return false;
5372
+ return true;
5373
+ }
5374
+ function describeMatcher(m) {
5375
+ if (typeof m === "string") return m;
5376
+ if (m instanceof RegExp) return `/${m.source}/${m.flags}`;
5377
+ return `/${m.$regex}/${m.flags}`;
5378
+ }
5379
+ function describePredicate(p) {
5380
+ const parts = [];
5381
+ if (p.name !== void 0) parts.push(`name=${describeMatcher(p.name)}`);
5382
+ if (p.tool !== void 0) parts.push(`tool=${describeMatcher(p.tool)}`);
5383
+ if (p.attr !== void 0) {
5384
+ for (const [k, v] of Object.entries(p.attr)) {
5385
+ parts.push(
5386
+ `attr.${k}=${v instanceof RegExp || isSerializedRegex(v) ? describeMatcher(v) : JSON.stringify(v)}`
5387
+ );
5388
+ }
5389
+ }
5390
+ if (typeof p.custom === "function") parts.push(`custom=${p.custom.name || "fn"}`);
5391
+ return parts.join(",");
5392
+ }
5393
+ function normalizeMatcher(m) {
5394
+ return m instanceof RegExp ? { $regex: m.source, flags: m.flags } : m;
5395
+ }
5396
+ function normalizePredicate(p, where) {
5397
+ assertPredicate(p, where);
5398
+ const out = {};
5399
+ if (p.name !== void 0) out.name = normalizeMatcher(p.name);
5400
+ if (p.tool !== void 0) out.tool = normalizeMatcher(p.tool);
5401
+ if (p.attr !== void 0) {
5402
+ out.attr = Object.fromEntries(
5403
+ Object.entries(p.attr).map(([k, v]) => [k, v instanceof RegExp ? normalizeMatcher(v) : v])
5404
+ );
5405
+ }
5406
+ if (typeof p.custom === "function") {
5407
+ out.custom = p.custom;
5408
+ out.requiresCustom = true;
5409
+ }
5410
+ return out;
5411
+ }
5412
+ var TraceContractBuilder = class {
5413
+ constructor(name) {
5414
+ this.name = name;
5415
+ }
5416
+ name;
5417
+ rules = [];
5418
+ /** Every span in the trace must satisfy `p`. */
5419
+ always(p, label) {
5420
+ const np = normalizePredicate(p, `traceContract("${this.name}").always`);
5421
+ return this.add({ kind: "always", label: label ?? `always(${describePredicate(np)})`, p: np });
5422
+ }
5423
+ /** No span in the trace may satisfy `p`. */
5424
+ never(p, label) {
5425
+ const np = normalizePredicate(p, `traceContract("${this.name}").never`);
5426
+ return this.add({ kind: "never", label: label ?? `never(${describePredicate(np)})`, p: np });
5427
+ }
5428
+ /** At least one span in the trace must satisfy `p`. */
5429
+ eventually(p, label) {
5430
+ const np = normalizePredicate(p, `traceContract("${this.name}").eventually`);
5431
+ return this.add({
5432
+ kind: "eventually",
5433
+ label: label ?? `eventually(${describePredicate(np)})`,
5434
+ p: np
5435
+ });
5436
+ }
5437
+ /** Every `b`-match must have a strictly earlier `a`-match. */
5438
+ precedes(a, b, label) {
5439
+ const na = normalizePredicate(a, `traceContract("${this.name}").precedes (a)`);
5440
+ const nb = normalizePredicate(b, `traceContract("${this.name}").precedes (b)`);
5441
+ return this.add({
5442
+ kind: "precedes",
5443
+ label: label ?? `precedes(${describePredicate(na)} -> ${describePredicate(nb)})`,
5444
+ a: na,
5445
+ b: nb
5446
+ });
5447
+ }
5448
+ /** Every `p`-match is a violation unless a strictly earlier `prior`-match exists. */
5449
+ neverUnless(p, prior, label) {
5450
+ const np = normalizePredicate(p, `traceContract("${this.name}").neverUnless (p)`);
5451
+ const nprior = normalizePredicate(prior, `traceContract("${this.name}").neverUnless (prior)`);
5452
+ return this.add({
5453
+ kind: "neverUnless",
5454
+ label: label ?? `neverUnless(${describePredicate(np)} unless ${describePredicate(nprior)})`,
5455
+ p: np,
5456
+ prior: nprior
5457
+ });
5458
+ }
5459
+ build() {
5460
+ if (this.rules.length === 0) {
5461
+ throw new ValidationError(
5462
+ `traceContract("${this.name}").build(): no rules \u2014 an empty contract would vacuously pass`
5463
+ );
5464
+ }
5465
+ return { name: this.name, rules: [...this.rules] };
5466
+ }
5467
+ add(rule) {
5468
+ let label = rule.label;
5469
+ let n = 2;
5470
+ while (this.rules.some((r) => r.label === label)) {
5471
+ label = `${rule.label} #${n}`;
5472
+ n += 1;
5473
+ }
5474
+ this.rules.push({ ...rule, label });
5475
+ return this;
5476
+ }
5477
+ };
5478
+ function traceContract(name) {
5479
+ if (typeof name !== "string" || name.length === 0) {
5480
+ throw new ValidationError("traceContract: name must be a non-empty string");
5481
+ }
5482
+ return new TraceContractBuilder(name);
5483
+ }
5484
+ function assertContract(contract) {
5485
+ if (typeof contract?.name !== "string" || contract.name.length === 0) {
5486
+ throw new ValidationError("evaluateTraceContract: contract.name must be a non-empty string");
5487
+ }
5488
+ if (!Array.isArray(contract.rules) || contract.rules.length === 0) {
5489
+ throw new ValidationError(
5490
+ `evaluateTraceContract: contract "${contract.name}" has no rules \u2014 an empty contract would vacuously pass`
5491
+ );
5492
+ }
5493
+ const seen = /* @__PURE__ */ new Set();
5494
+ for (const rule of contract.rules) {
5495
+ const where = `contract "${contract.name}" rule "${rule?.label ?? "<unlabeled>"}"`;
5496
+ if (typeof rule?.label !== "string" || rule.label.length === 0) {
5497
+ throw new ValidationError(`${where}: label must be a non-empty string`);
5498
+ }
5499
+ if (seen.has(rule.label)) {
5500
+ throw new ValidationError(`${where}: duplicate label would collapse per-rule scores`);
5501
+ }
5502
+ seen.add(rule.label);
5503
+ switch (rule.kind) {
5504
+ case "always":
5505
+ case "never":
5506
+ case "eventually":
5507
+ assertPredicate(rule.p, `${where} (p)`);
5508
+ break;
5509
+ case "precedes":
5510
+ assertPredicate(rule.a, `${where} (a)`);
5511
+ assertPredicate(rule.b, `${where} (b)`);
5512
+ break;
5513
+ case "neverUnless":
5514
+ assertPredicate(rule.p, `${where} (p)`);
5515
+ assertPredicate(rule.prior, `${where} (prior)`);
5516
+ break;
5517
+ default:
5518
+ throw new ValidationError(
5519
+ `${where}: unknown rule kind "${String(rule.kind)}"`
5520
+ );
5521
+ }
5522
+ }
5523
+ }
5524
+ function orderSpans(spans) {
5525
+ const timed = spans.filter(
5526
+ (s) => typeof s.startedAt === "number" && Number.isFinite(s.startedAt)
5527
+ ).length;
5528
+ if (timed === spans.length) {
5529
+ return spans.map((span, i) => ({ span, i })).sort((x, y) => x.span.startedAt - y.span.startedAt || x.i - y.i).map((x) => x.span);
5530
+ }
5531
+ if (timed === 0) return [...spans];
5532
+ throw new ValidationError(
5533
+ `evaluateTraceContract: ${timed}/${spans.length} spans carry a finite startedAt \u2014 mixed timestamps make ordering ambiguous; stamp every span or none`
5534
+ );
5535
+ }
5536
+ function spanRef(span, index) {
5537
+ return span.spanId ?? `#${index}`;
5538
+ }
5539
+ function checkGuarded(args) {
5540
+ const out = [];
5541
+ let guardSeen = false;
5542
+ for (let i = 0; i < args.ordered.length; i++) {
5543
+ const span = args.ordered[i];
5544
+ if (!guardSeen && matchSpan(span, args.guarded)) {
5545
+ out.push({ rule: args.label, spanId: span.spanId, detail: args.detail(spanRef(span, i)) });
5546
+ }
5547
+ if (matchSpan(span, args.guard)) guardSeen = true;
5548
+ }
5549
+ return out;
5550
+ }
5551
+ function checkRule(rule, ordered) {
5552
+ switch (rule.kind) {
5553
+ case "always": {
5554
+ const out = [];
5555
+ for (let i = 0; i < ordered.length; i++) {
5556
+ const span = ordered[i];
5557
+ if (!matchSpan(span, rule.p)) {
5558
+ out.push({
5559
+ rule: rule.label,
5560
+ spanId: span.spanId,
5561
+ detail: `span ${spanRef(span, i)} ("${span.name ?? ""}") fails always(${describePredicate(rule.p)})`
5562
+ });
5563
+ }
5564
+ }
5565
+ return out;
5566
+ }
5567
+ case "never": {
5568
+ const out = [];
5569
+ for (let i = 0; i < ordered.length; i++) {
5570
+ const span = ordered[i];
5571
+ if (matchSpan(span, rule.p)) {
5572
+ out.push({
5573
+ rule: rule.label,
5574
+ spanId: span.spanId,
5575
+ detail: `span ${spanRef(span, i)} ("${span.name ?? ""}") matches never(${describePredicate(rule.p)})`
5576
+ });
5577
+ }
5578
+ }
5579
+ return out;
5580
+ }
5581
+ case "eventually": {
5582
+ const hit = ordered.some((span) => matchSpan(span, rule.p));
5583
+ return hit ? [] : [
5584
+ {
5585
+ rule: rule.label,
5586
+ detail: `no span matches eventually(${describePredicate(rule.p)}) over ${ordered.length} span(s)`
5587
+ }
5588
+ ];
5589
+ }
5590
+ case "precedes":
5591
+ return checkGuarded({
5592
+ label: rule.label,
5593
+ ordered,
5594
+ guard: rule.a,
5595
+ guarded: rule.b,
5596
+ detail: (ref) => `span ${ref} matches ${describePredicate(rule.b)} with no earlier ${describePredicate(rule.a)} match`
5597
+ });
5598
+ case "neverUnless":
5599
+ return checkGuarded({
5600
+ label: rule.label,
5601
+ ordered,
5602
+ guard: rule.prior,
5603
+ guarded: rule.p,
5604
+ detail: (ref) => `span ${ref} matches ${describePredicate(rule.p)} with no earlier ${describePredicate(rule.prior)} match`
5605
+ });
5606
+ }
4989
5607
  }
4990
- function keyPreserved(key) {
5608
+ function evaluateTraceContract(contract, spans) {
5609
+ assertContract(contract);
5610
+ const ordered = orderSpans(spans);
5611
+ const scores2 = {};
5612
+ const violations = [];
5613
+ for (const rule of contract.rules) {
5614
+ const ruleViolations = checkRule(rule, ordered);
5615
+ scores2[rule.label] = ruleViolations.length === 0 ? 1 : 0;
5616
+ violations.push(...ruleViolations);
5617
+ }
5618
+ const ruleCount = contract.rules.length;
5619
+ const passCount = Object.values(scores2).filter((s) => s === 1).length;
4991
5620
  return {
4992
- id: `preserved(${key})`,
4993
- description: `"${key}" unchanged from before to after`,
4994
- score: ({ before, after }) => before[key] !== void 0 && before[key] === after[key] ? 1 : 0
5621
+ contract: contract.name,
5622
+ valid: passCount === ruleCount,
5623
+ score: passCount / ruleCount,
5624
+ scores: scores2,
5625
+ violations,
5626
+ notes: `${passCount}/${ruleCount} rules passed`
4995
5627
  };
4996
5628
  }
4997
- function collectionPreserved(key, minRatio = 1) {
4998
- return {
4999
- id: `collection-preserved(${key})`,
5000
- description: `"${key}" length \u2265 ${minRatio} \xD7 prior length`,
5001
- score: ({ before, after }) => {
5002
- const b = before[key];
5003
- const a = after[key];
5004
- if (!Array.isArray(b) || !Array.isArray(a)) return 0;
5005
- if (b.length === 0) return a.length === 0 ? 1 : 1;
5006
- return Math.min(1, a.length / (b.length * minRatio));
5007
- }
5008
- };
5629
+ function checkTraceContracts(spans, contracts) {
5630
+ if (contracts.length === 0) {
5631
+ throw new ValidationError(
5632
+ "checkTraceContracts: empty contract list would vacuously pass \u2014 supply at least one contract"
5633
+ );
5634
+ }
5635
+ const verdicts = contracts.map((c) => evaluateTraceContract(c, spans));
5636
+ return { verdicts, allValid: verdicts.every((v) => v.valid) };
5009
5637
  }
5010
- function statusAdvanced(key, progression) {
5638
+ function contractJudge(contracts, opts) {
5639
+ if (contracts.length === 0) {
5640
+ throw new ValidationError("contractJudge: at least one contract required");
5641
+ }
5642
+ for (const c of contracts) assertContract(c);
5643
+ const names = /* @__PURE__ */ new Set();
5644
+ for (const c of contracts) {
5645
+ if (names.has(c.name)) {
5646
+ throw new ValidationError(
5647
+ `contractJudge: duplicate contract name "${c.name}" would collapse judge dimensions`
5648
+ );
5649
+ }
5650
+ names.add(c.name);
5651
+ }
5652
+ if (typeof opts?.spans !== "function") {
5653
+ throw new ValidationError("contractJudge: opts.spans extraction function is required");
5654
+ }
5655
+ const dimensions = contracts.map((c) => ({
5656
+ key: c.name,
5657
+ description: c.rules.map((r) => r.label).join("; ")
5658
+ }));
5011
5659
  return {
5012
- id: `status-advanced(${key})`,
5013
- description: `"${key}" progressed along ${progression.join("\u2192")}`,
5014
- score: ({ before, after }) => {
5015
- const bi = progression.indexOf(String(before[key]));
5016
- const ai2 = progression.indexOf(String(after[key]));
5017
- if (bi === -1 || ai2 === -1) return 0;
5018
- return ai2 >= bi ? 1 : 0;
5660
+ name: opts.name ?? "trace-contracts",
5661
+ dimensions,
5662
+ score({ artifact, scenario }) {
5663
+ const spans = opts.spans({ artifact, scenario });
5664
+ if (!Array.isArray(spans)) {
5665
+ throw new ValidationError(
5666
+ `contractJudge: spans() must return a span array, got ${typeof spans}`
5667
+ );
5668
+ }
5669
+ const { verdicts } = checkTraceContracts(spans, contracts);
5670
+ const dims = {};
5671
+ const violations = [];
5672
+ for (const v of verdicts) {
5673
+ dims[v.contract] = v.score;
5674
+ violations.push(...v.violations);
5675
+ }
5676
+ const composite = verdicts.reduce((acc, v) => acc + v.score, 0) / verdicts.length;
5677
+ return {
5678
+ dimensions: dims,
5679
+ composite,
5680
+ notes: violations.length === 0 ? "all trace contracts satisfied" : violations.slice(0, 8).map((x) => `${x.rule}: ${x.detail}`).join("\n")
5681
+ };
5019
5682
  }
5020
5683
  };
5021
5684
  }
5022
5685
 
5023
- // src/ui-finding.ts
5024
- var UI_LENSES = [
5025
- "consistency",
5026
- "hierarchy",
5027
- "layout",
5028
- "ux-flow",
5029
- "duplication",
5030
- "accessibility",
5031
- "responsive",
5032
- "states",
5033
- "content",
5034
- "interaction",
5035
- "performance-perceived",
5036
- "other"
5037
- ];
5038
- var UI_FINDING_SEVERITIES = [
5039
- "critical",
5040
- "high",
5041
- "med",
5042
- "low"
5043
- ];
5044
-
5045
5686
  // src/behavior-dsl.ts
5046
5687
  var BehaviorAssertion = class {
5047
5688
  constructor(store, runId) {
@@ -5109,6 +5750,25 @@ var BehaviorAssertion = class {
5109
5750
  }
5110
5751
  };
5111
5752
  }
5753
+ /** Evaluate a finite-trace temporal contract (`traceContract(...)`) over
5754
+ * this run's span sequence. See `trace-contracts.ts` for the operators. */
5755
+ toSatisfyContract(contract) {
5756
+ return {
5757
+ label: `agent(${this.runId}).toSatisfyContract(${contract.name})`,
5758
+ check: async () => {
5759
+ const spans = await this.store.spans({ runId: this.runId });
5760
+ const verdict = evaluateTraceContract(contract, spans);
5761
+ if (verdict.valid) {
5762
+ return { ok: true, detail: `contract "${contract.name}": ${verdict.notes}` };
5763
+ }
5764
+ return {
5765
+ ok: false,
5766
+ detail: verdict.violations.map((v) => `${v.rule}: ${v.detail}`).join("; "),
5767
+ evidence: verdict.violations.find((v) => v.spanId)?.spanId
5768
+ };
5769
+ }
5770
+ };
5771
+ }
5112
5772
  toNeverCall(toolName) {
5113
5773
  return {
5114
5774
  label: `agent(${this.runId}).toNeverCall(${toolName})`,
@@ -5711,90 +6371,6 @@ async function promptBisect(options) {
5711
6371
  };
5712
6372
  }
5713
6373
 
5714
- // src/counterfactual.ts
5715
- async function runCounterfactual(store, originalRunId, mutation, runner) {
5716
- const originalRun = await store.getRun(originalRunId);
5717
- if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
5718
- const trajectory = await buildTrajectory(store, originalRunId);
5719
- if (mutation.at < 0 || mutation.at >= trajectory.steps.length) {
5720
- throw new ValidationError(
5721
- `counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`
5722
- );
5723
- }
5724
- const targetStep = trajectory.steps[mutation.at];
5725
- const mutatedStep = applyMutation(targetStep, mutation);
5726
- const cfEmitter = new TraceEmitter(store);
5727
- await cfEmitter.startRun({
5728
- scenarioId: originalRun.scenarioId,
5729
- variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
5730
- projectId: originalRun.projectId,
5731
- parentRunId: originalRunId,
5732
- layer: "meta",
5733
- tags: { counterfactual: "true", mutationKind: mutation.kind, mutationAt: String(mutation.at) }
5734
- });
5735
- await runner.executeFrom(
5736
- {
5737
- originalRunId,
5738
- originalTrajectory: trajectory,
5739
- prefix: trajectory.steps.slice(0, mutation.at),
5740
- mutation,
5741
- mutatedStep
5742
- },
5743
- cfEmitter
5744
- );
5745
- const counterfactual = await store.getRun(cfEmitter.runId);
5746
- const delta = {
5747
- originalOutcomeScore: originalRun.outcome?.score ?? null,
5748
- counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
5749
- deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
5750
- };
5751
- return { counterfactualRunId: cfEmitter.runId, originalRunId, mutation, delta };
5752
- }
5753
- function applyMutation(step, mutation) {
5754
- if (mutation.kind === "swap-model" && step.span.kind === "llm") {
5755
- const llm = step.span;
5756
- return { ...step, span: { ...llm, model: mutation.newModel } };
5757
- }
5758
- if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
5759
- const tool = step.span;
5760
- return { ...step, span: { ...tool, result: mutation.newResult } };
5761
- }
5762
- if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
5763
- const llm = step.span;
5764
- return {
5765
- ...step,
5766
- span: {
5767
- ...llm,
5768
- messages: [{ role: "system", content: mutation.content }, ...llm.messages]
5769
- }
5770
- };
5771
- }
5772
- if (mutation.kind === "custom") return mutation.apply(step);
5773
- return step;
5774
- }
5775
- function attributeCounterfactuals(results) {
5776
- const grouped = /* @__PURE__ */ new Map();
5777
- for (const r of results) {
5778
- const arr = grouped.get(r.mutation.kind) ?? [];
5779
- arr.push(r);
5780
- grouped.set(r.mutation.kind, arr);
5781
- }
5782
- const out = [];
5783
- for (const [kind, items] of grouped) {
5784
- const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
5785
- if (deltas.length === 0) continue;
5786
- const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
5787
- const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
5788
- out.push({
5789
- mutationKind: kind,
5790
- n: deltas.length,
5791
- meanAbsDelta: meanAbs,
5792
- meanSignedDelta: meanSigned
5793
- });
5794
- }
5795
- return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
5796
- }
5797
-
5798
6374
  // src/cross-trace-diff.ts
5799
6375
  async function crossTraceDiff(store, runA, runB, options = {}) {
5800
6376
  const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
@@ -6986,14 +7562,18 @@ function aggregate(layers, results, startedAt, startedAtMs) {
6986
7562
  }
6987
7563
  }
6988
7564
  const finishedAtMs = Date.now();
7565
+ const allPass = ranAnyScoredLayer && !anyScoredLayerFailed && failCount === 0 && errorCount === 0;
7566
+ const blendedScore = scoredWeightSum > 0 ? scoredWeightedTotal / scoredWeightSum : 0;
6989
7567
  return {
6990
7568
  layers: results,
6991
7569
  passCount,
6992
7570
  failCount,
6993
7571
  skippedCount,
6994
7572
  errorCount,
6995
- allPass: ranAnyScoredLayer && !anyScoredLayerFailed && failCount === 0 && errorCount === 0,
6996
- blendedScore: scoredWeightSum > 0 ? scoredWeightedTotal / scoredWeightSum : 0,
7573
+ allPass,
7574
+ blendedScore,
7575
+ valid: allPass,
7576
+ score: blendedScore,
6997
7577
  durationMs: finishedAtMs - startedAtMs,
6998
7578
  startedAt,
6999
7579
  finishedAt: new Date(finishedAtMs).toISOString()
@@ -8208,69 +8788,6 @@ function fmt(x) {
8208
8788
  return x.toFixed(4);
8209
8789
  }
8210
8790
 
8211
- // src/judge-retry.ts
8212
- var DEFAULT_MAX_ATTEMPTS = 3;
8213
- var DEFAULT_TIMEOUT_MS = 3e5;
8214
- function sleep(ms) {
8215
- return new Promise((resolve) => setTimeout(resolve, ms));
8216
- }
8217
- async function withJudgeRetry(judgeFn, policy = {}) {
8218
- const maxAttempts = policy.maxAttempts ?? DEFAULT_MAX_ATTEMPTS;
8219
- const timeoutMs = policy.timeoutMs ?? DEFAULT_TIMEOUT_MS;
8220
- const backoff = policy.backoffMs ?? backoffMs;
8221
- const isRetryable = policy.isRetryable ?? isTransientLlmError;
8222
- const models = policy.models && policy.models.length > 0 ? policy.models : [void 0];
8223
- let totalAttempts = 0;
8224
- const attemptErrors = [];
8225
- let lastError;
8226
- for (const model of models) {
8227
- for (let attempt = 0; attempt < maxAttempts; attempt++) {
8228
- totalAttempts += 1;
8229
- const controller = new AbortController();
8230
- const timer = setTimeout(() => controller.abort(new Error("TimeoutError")), timeoutMs);
8231
- try {
8232
- const value = await judgeFn(model, controller.signal);
8233
- clearTimeout(timer);
8234
- return {
8235
- value,
8236
- succeeded: true,
8237
- attempts: totalAttempts,
8238
- modelUsed: model,
8239
- attemptErrors
8240
- };
8241
- } catch (err) {
8242
- clearTimeout(timer);
8243
- const errObj = err instanceof Error ? err : new Error(String(err));
8244
- lastError = errObj;
8245
- attemptErrors.push({
8246
- attempt: totalAttempts,
8247
- model: model ?? "(default)",
8248
- error: errObj.message
8249
- });
8250
- if (!isRetryable(errObj)) {
8251
- return {
8252
- value: null,
8253
- succeeded: false,
8254
- attempts: totalAttempts,
8255
- error: errObj,
8256
- attemptErrors
8257
- };
8258
- }
8259
- if (attempt < maxAttempts - 1) {
8260
- await sleep(backoff(attempt));
8261
- }
8262
- }
8263
- }
8264
- }
8265
- return {
8266
- value: null,
8267
- succeeded: false,
8268
- attempts: totalAttempts,
8269
- error: lastError,
8270
- attemptErrors
8271
- };
8272
- }
8273
-
8274
8791
  // src/orthogonality.ts
8275
8792
  function passOrthogonality(input) {
8276
8793
  const passes = input.passes;
@@ -8649,6 +9166,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
8649
9166
  });
8650
9167
  try {
8651
9168
  const allScores = [];
9169
+ let failedJudges = 0;
8652
9170
  for (let i = 0; i < judges.length; i++) {
8653
9171
  const judge = judges[i];
8654
9172
  const name = judgeNames[i] ?? `judge_${i}`;
@@ -8656,8 +9174,13 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
8656
9174
  emitter: opts.emitter,
8657
9175
  parentSpanId: ensembleSpan.span.spanId
8658
9176
  });
8659
- const scores2 = await tracedFn(tc, input);
8660
- allScores.push(...scores2);
9177
+ try {
9178
+ const scores2 = await tracedFn(tc, input);
9179
+ allScores.push(...scores2);
9180
+ } catch (err) {
9181
+ if (!(err instanceof JudgeParseError)) throw err;
9182
+ failedJudges++;
9183
+ }
8661
9184
  }
8662
9185
  const composite = allScores.length > 0 ? allScores.reduce((sum3, s) => sum3 + s.score, 0) / allScores.length : 0;
8663
9186
  await ensembleSpan.end({
@@ -8665,6 +9188,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
8665
9188
  "judge.ensemble_size": judges.length,
8666
9189
  "judge.composite_score": composite,
8667
9190
  "judge.total_dimensions": allScores.length,
9191
+ "judge.failed_judges": failedJudges,
8668
9192
  "eval.phase": "judge"
8669
9193
  }
8670
9194
  });
@@ -9094,8 +9618,256 @@ function applyDomainPatch(p, sectionId, newBody) {
9094
9618
  function sectionHash(section) {
9095
9619
  return surfaceContentHash(JSON.stringify({ title: section.title, body: section.body }));
9096
9620
  }
9621
+
9622
+ // src/cost-report.ts
9623
+ function costReport(ledger) {
9624
+ const summary = ledger.summary();
9625
+ const perModel = /* @__PURE__ */ new Map();
9626
+ for (const entry of ledger.list()) {
9627
+ const roll = perModel.get(entry.model) ?? {
9628
+ model: entry.model,
9629
+ usd: 0,
9630
+ entries: 0,
9631
+ unpriced: false
9632
+ };
9633
+ roll.usd += entry.costUsd;
9634
+ roll.entries += 1;
9635
+ if (entry.costUnknown) roll.unpriced = true;
9636
+ perModel.set(entry.model, roll);
9637
+ }
9638
+ return {
9639
+ perChannel: summary.byChannel,
9640
+ total: {
9641
+ usd: summary.totalCostUsd,
9642
+ unknownEntries: summary.byChannel.reduce((sum3, c) => sum3 + c.unpricedCalls, 0)
9643
+ },
9644
+ perModel: [...perModel.values()].sort((a, b) => a.model.localeCompare(b.model))
9645
+ };
9646
+ }
9647
+ function attachCostToReport(report, ledger) {
9648
+ if ("cost" in report) {
9649
+ throw new ValidationError(
9650
+ "attachCostToReport: report already has a 'cost' key \u2014 refusing to overwrite an existing stamp"
9651
+ );
9652
+ }
9653
+ return { ...report, cost: costReport(ledger) };
9654
+ }
9655
+
9656
+ // src/model-seats.ts
9657
+ var seatPresets = {
9658
+ economy: {
9659
+ worker: "kimi-k2.6",
9660
+ judges: ["kimi-k2.6", "deepseek-v4-pro", "gpt-4.1-mini"],
9661
+ analyst: "gpt-4.1-mini",
9662
+ reflection: "gpt-4.1-mini",
9663
+ verifier: "deepseek-v4-pro"
9664
+ },
9665
+ frontier: {}
9666
+ };
9667
+ var SeatUnsetError = class extends ConfigError {
9668
+ constructor(seat) {
9669
+ super(
9670
+ `ModelSeats: seat '${seat}' is unset and no fallback was given \u2014 name a model explicitly (a model id is a budget decision, never a silent default)`
9671
+ );
9672
+ this.seat = seat;
9673
+ }
9674
+ seat;
9675
+ };
9676
+ function resolveSeat(seats, seat, fallback) {
9677
+ const value = seats[seat];
9678
+ if (seat === "judges") {
9679
+ if (value !== void 0 && !Array.isArray(value)) {
9680
+ throw new ValidationError(`ModelSeats: seat 'judges' must be a string[], got ${typeof value}`);
9681
+ }
9682
+ const models = Array.isArray(value) ? value : [];
9683
+ if (models.length > 0) {
9684
+ const blank = models.findIndex((m) => typeof m !== "string" || m.trim() === "");
9685
+ if (blank >= 0) {
9686
+ throw new ValidationError(
9687
+ `ModelSeats: judges[${blank}] is blank \u2014 every panel model must be a non-empty id`
9688
+ );
9689
+ }
9690
+ return [...models];
9691
+ }
9692
+ } else {
9693
+ if (value !== void 0 && typeof value !== "string") {
9694
+ throw new ValidationError(`ModelSeats: seat '${seat}' must be a string, got ${typeof value}`);
9695
+ }
9696
+ if (typeof value === "string" && value.trim() !== "") return value;
9697
+ }
9698
+ if (fallback !== void 0) {
9699
+ if (fallback.trim() === "") {
9700
+ throw new ValidationError(`ModelSeats: fallback for seat '${seat}' is blank`);
9701
+ }
9702
+ return seat === "judges" ? [fallback] : fallback;
9703
+ }
9704
+ throw new SeatUnsetError(seat);
9705
+ }
9706
+
9707
+ // src/verdict-cache.ts
9708
+ import { createHash } from "crypto";
9709
+ import { appendFileSync as appendFileSync3, existsSync as existsSync5, readFileSync as readFileSync6 } from "fs";
9710
+ function canonicalizeAt(value, path) {
9711
+ if (value === null) return "null";
9712
+ switch (typeof value) {
9713
+ case "boolean":
9714
+ return value ? "true" : "false";
9715
+ case "number":
9716
+ if (!Number.isFinite(value)) {
9717
+ throw new Error(
9718
+ `canonicalJson: non-finite number (${value}) at ${path} \u2014 ambiguity is an error, not a coercion`
9719
+ );
9720
+ }
9721
+ return JSON.stringify(value);
9722
+ case "string":
9723
+ return JSON.stringify(value);
9724
+ case "undefined":
9725
+ case "function":
9726
+ case "symbol":
9727
+ throw new Error(
9728
+ `canonicalJson: ${typeof value} at ${path} \u2014 ambiguity is an error, not a coercion`
9729
+ );
9730
+ case "bigint":
9731
+ throw new Error(`canonicalJson: bigint at ${path} \u2014 not representable in JSON`);
9732
+ case "object":
9733
+ break;
9734
+ }
9735
+ const obj = value;
9736
+ if (typeof obj["toJSON"] === "function") {
9737
+ return canonicalizeAt(obj.toJSON(), path);
9738
+ }
9739
+ if (Array.isArray(obj)) {
9740
+ return `[${obj.map((item, i) => canonicalizeAt(item, `${path}[${i}]`)).join(",")}]`;
9741
+ }
9742
+ if (obj instanceof Map || obj instanceof Set) {
9743
+ throw new Error(
9744
+ `canonicalJson: ${obj instanceof Map ? "Map" : "Set"} at ${path} \u2014 would serialize as '{}'; convert to a plain object/array first`
9745
+ );
9746
+ }
9747
+ const keys = Object.keys(obj).sort();
9748
+ const parts = keys.map((k) => `${JSON.stringify(k)}:${canonicalizeAt(obj[k], `${path}.${k}`)}`);
9749
+ return `{${parts.join(",")}}`;
9750
+ }
9751
+ function canonicalJson(value) {
9752
+ return canonicalizeAt(value, "$");
9753
+ }
9754
+ function contentHash(value) {
9755
+ return createHash("sha256").update(canonicalJson(value)).digest("hex");
9756
+ }
9757
+ function inMemoryVerdictCache() {
9758
+ const entries = /* @__PURE__ */ new Map();
9759
+ return {
9760
+ get: (key) => entries.get(key),
9761
+ set: (key, score) => {
9762
+ entries.set(key, score);
9763
+ }
9764
+ };
9765
+ }
9766
+ function parseCacheLine(line, path, lineNo) {
9767
+ let parsed;
9768
+ try {
9769
+ parsed = JSON.parse(line);
9770
+ } catch (err) {
9771
+ throw new Error(
9772
+ `fileVerdictCache: corrupt JSONL at ${path}:${lineNo} \u2014 ${err instanceof Error ? err.message : String(err)}`
9773
+ );
9774
+ }
9775
+ const rec = parsed;
9776
+ if (typeof rec !== "object" || rec === null || typeof rec.key !== "string" || typeof rec.score !== "object" || rec.score === null || typeof rec.score.composite !== "number" || typeof rec.score.dimensions !== "object") {
9777
+ throw new Error(
9778
+ `fileVerdictCache: invalid record shape at ${path}:${lineNo} \u2014 expected {key, score:{dimensions, composite, notes}}`
9779
+ );
9780
+ }
9781
+ return rec;
9782
+ }
9783
+ function fileVerdictCache(path) {
9784
+ const entries = /* @__PURE__ */ new Map();
9785
+ if (existsSync5(path)) {
9786
+ const lines = readFileSync6(path, "utf8").split("\n");
9787
+ for (let i = 0; i < lines.length; i++) {
9788
+ const line = lines[i];
9789
+ if (line === void 0 || line.trim() === "") continue;
9790
+ const rec = parseCacheLine(line, path, i + 1);
9791
+ entries.set(rec.key, rec.score);
9792
+ }
9793
+ }
9794
+ return {
9795
+ get: (key) => entries.get(key),
9796
+ set: (key, score) => {
9797
+ appendFileSync3(path, `${JSON.stringify({ key, score })}
9798
+ `, "utf8");
9799
+ entries.set(key, score);
9800
+ }
9801
+ };
9802
+ }
9803
+ function cachedJudge(judge, store, options) {
9804
+ if (typeof options.judgeVersion !== "string" || options.judgeVersion.trim() === "") {
9805
+ throw new Error("cachedJudge: judgeVersion is required and must be a non-empty string");
9806
+ }
9807
+ const stats = { hits: 0, misses: 0 };
9808
+ const wrapped = {
9809
+ name: judge.name,
9810
+ dimensions: judge.dimensions,
9811
+ async score(input) {
9812
+ const key = contentHash({
9813
+ artifact: canonicalJson(input.artifact),
9814
+ scenarioId: input.scenario.id,
9815
+ judgeName: judge.name,
9816
+ dimensions: judge.dimensions,
9817
+ judgeVersion: options.judgeVersion
9818
+ });
9819
+ const cached = await store.get(key);
9820
+ if (cached !== void 0) {
9821
+ stats.hits += 1;
9822
+ return cached;
9823
+ }
9824
+ const score = await judge.score(input);
9825
+ await store.set(key, score);
9826
+ stats.misses += 1;
9827
+ return score;
9828
+ },
9829
+ stats: () => ({ ...stats })
9830
+ };
9831
+ if (judge.appliesTo) wrapped.appliesTo = judge.appliesTo;
9832
+ return wrapped;
9833
+ }
9834
+
9835
+ // src/attestation.ts
9836
+ var ATTESTATION_ALGORITHM = "sha256/canonical-json";
9837
+ function attest(report, provenance) {
9838
+ return {
9839
+ reportHash: contentHash(report),
9840
+ provenance,
9841
+ algorithm: ATTESTATION_ALGORITHM
9842
+ };
9843
+ }
9844
+ function verifyAttestation(report, attested) {
9845
+ if (attested.algorithm !== ATTESTATION_ALGORITHM) {
9846
+ return {
9847
+ valid: false,
9848
+ reason: `unknown algorithm '${attested.algorithm}' \u2014 this verifier only checks '${ATTESTATION_ALGORITHM}'`
9849
+ };
9850
+ }
9851
+ let recomputed;
9852
+ try {
9853
+ recomputed = contentHash(report);
9854
+ } catch (err) {
9855
+ return {
9856
+ valid: false,
9857
+ reason: `report is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
9858
+ };
9859
+ }
9860
+ if (recomputed !== attested.reportHash) {
9861
+ return {
9862
+ valid: false,
9863
+ reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
9864
+ };
9865
+ }
9866
+ return { valid: true };
9867
+ }
9097
9868
  export {
9098
9869
  AGENT_PROFILE_KINDS,
9870
+ ATTESTATION_ALGORITHM,
9099
9871
  AgentDriver,
9100
9872
  AgentEvalError,
9101
9873
  AgentProfileCellValidationError,
@@ -9150,6 +9922,7 @@ export {
9150
9922
  InMemoryTraceStore,
9151
9923
  InMemoryWorkspaceInspector,
9152
9924
  JudgeError,
9925
+ JudgeParseError,
9153
9926
  JudgeRunner,
9154
9927
  KNOWLEDGE_GAP_KIND_SPEC,
9155
9928
  KNOWLEDGE_POISONING_KIND_SPEC,
@@ -9182,6 +9955,7 @@ export {
9182
9955
  SKILL_USAGE_ANALYST,
9183
9956
  SandboxHarness,
9184
9957
  ScenarioRegistry,
9958
+ SeatUnsetError,
9185
9959
  SingleBackendError,
9186
9960
  SkillUsageAnalyst,
9187
9961
  SpanNotFoundError,
@@ -9192,6 +9966,7 @@ export {
9192
9966
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
9193
9967
  TRACE_SCHEMA_VERSION,
9194
9968
  TokenCounter,
9969
+ TraceContractBuilder,
9195
9970
  TraceEmitter,
9196
9971
  TraceFileMissingError,
9197
9972
  TraceNotFoundError,
@@ -9227,6 +10002,8 @@ export {
9227
10002
  assertSingleBackend,
9228
10003
  assignFeedbackSplit,
9229
10004
  assignHeldOutTag,
10005
+ attachCostToReport,
10006
+ attest,
9230
10007
  attributeCounterfactuals,
9231
10008
  backoffMs,
9232
10009
  deterministicSplit as benchmarkDeterministicSplit,
@@ -9248,17 +10025,20 @@ export {
9248
10025
  buildTraceInsightPrompt,
9249
10026
  buildTrajectory,
9250
10027
  byteLengthRange,
10028
+ cachedJudge,
9251
10029
  calibrateJudge,
9252
10030
  calibrateJudgeContinuous,
9253
10031
  callLlm,
9254
10032
  callLlmJson,
9255
10033
  canaryLeakView,
10034
+ canonicalJson,
9256
10035
  canonicalize,
9257
10036
  captureFetchToRawSink,
9258
10037
  causalAttribution,
9259
10038
  checkBehavioralCanary,
9260
10039
  checkCanaries,
9261
10040
  checkSlos,
10041
+ checkTraceContracts,
9262
10042
  clamp01,
9263
10043
  classifyEuAiRisk,
9264
10044
  classifyFailure,
@@ -9272,6 +10052,7 @@ export {
9272
10052
  compareReferenceReplay,
9273
10053
  compareToBaseline,
9274
10054
  compilerJudge,
10055
+ completionVerdict,
9275
10056
  composeParsers,
9276
10057
  composeValidators,
9277
10058
  computeExperimentStats,
@@ -9280,7 +10061,9 @@ export {
9280
10061
  computeTraceMetrics,
9281
10062
  confidenceInterval,
9282
10063
  containsAll,
10064
+ contentHash,
9283
10065
  continuousAgreement,
10066
+ contractJudge,
9284
10067
  controlFailureClassFromVerification,
9285
10068
  controlRunToFeedbackTrajectory,
9286
10069
  controlRunToRunRecord,
@@ -9288,6 +10071,7 @@ export {
9288
10071
  corpusInterRaterAgreement,
9289
10072
  corpusInterRaterAgreementFromJudgeScores,
9290
10073
  costForUsage,
10074
+ costReport,
9291
10075
  createAnalystAi,
9292
10076
  createAntiSlopJudge,
9293
10077
  createChatClient,
@@ -9326,6 +10110,8 @@ export {
9326
10110
  distillPlaybook,
9327
10111
  domainEvidencePattern,
9328
10112
  dominates,
10113
+ eProcess,
10114
+ ensembleJudge,
9329
10115
  estimateCost,
9330
10116
  estimateTokens,
9331
10117
  euAiActReport,
@@ -9335,6 +10121,7 @@ export {
9335
10121
  evaluateInterimReleaseConfidence,
9336
10122
  evaluateOracles,
9337
10123
  evaluateReleaseConfidence,
10124
+ evaluateTraceContract,
9338
10125
  executeScenario,
9339
10126
  expectAgent,
9340
10127
  exportRewardModel,
@@ -9354,6 +10141,7 @@ export {
9354
10141
  fileContains,
9355
10142
  fileExists,
9356
10143
  fileExperimentStore,
10144
+ fileVerdictCache,
9357
10145
  findAutoMatchNoExpectation,
9358
10146
  findConstructorCwdDropped,
9359
10147
  findFallbackToPass,
@@ -9386,6 +10174,7 @@ export {
9386
10174
  inMemoryReferenceReplayStore,
9387
10175
  inMemoryReviewStore,
9388
10176
  inMemoryRunRecordBackend,
10177
+ inMemoryVerdictCache,
9389
10178
  inferDomainKeywords,
9390
10179
  inferOtlpKind,
9391
10180
  interRaterReliability,
@@ -9420,13 +10209,16 @@ export {
9420
10209
  loadScorerFromGrader,
9421
10210
  localCommandRunner,
9422
10211
  lowercaseMutator,
10212
+ makeEvalTools,
9423
10213
  makeFinding,
9424
10214
  mannWhitneyU,
9425
10215
  matchGoldens,
10216
+ matchSpan,
9426
10217
  mergeLayerResults,
9427
10218
  mergeSteeringBundle,
9428
10219
  modelDescriptionBits,
9429
10220
  modelPriceKey,
10221
+ mulberry32,
9430
10222
  multiToolchainLayer,
9431
10223
  nistAiRmfReport,
9432
10224
  normalizeScores,
@@ -9495,6 +10287,7 @@ export {
9495
10287
  researchReport,
9496
10288
  resetLockedAppendersForTesting,
9497
10289
  resolveModelPricing,
10290
+ resolveSeat,
9498
10291
  roundTripRunRecord,
9499
10292
  rowCount,
9500
10293
  rowWhere,
@@ -9532,6 +10325,7 @@ export {
9532
10325
  scoreRedTeamOutput,
9533
10326
  scoreReferenceReplay,
9534
10327
  scoreTraceInsightReadiness,
10328
+ seatPresets,
9535
10329
  securityJudge,
9536
10330
  selectHarnessVariant,
9537
10331
  selfPreference,
@@ -9557,12 +10351,14 @@ export {
9557
10351
  throwIfRunIncomplete,
9558
10352
  toAgentProfileJson,
9559
10353
  toLangfuseEnvelope,
10354
+ toOpenAiTool,
9560
10355
  toPrometheusText,
9561
10356
  tokenizeDomainWords,
9562
10357
  toolNamesForRun,
9563
10358
  toolSpans,
9564
10359
  traceAnalystFunctionGroup,
9565
10360
  traceAnalystOnRunComplete,
10361
+ traceContract,
9566
10362
  traceJudge,
9567
10363
  traceJudgeEnsemble,
9568
10364
  tracedAnalyzeTraces,
@@ -9573,6 +10369,7 @@ export {
9573
10369
  validateRunRecord,
9574
10370
  verbosityBias,
9575
10371
  verifyAgentProfileCell,
10372
+ verifyAttestation,
9576
10373
  verifyCompletion,
9577
10374
  verifyManifest,
9578
10375
  visualDiff,