@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/index.js CHANGED
@@ -9,11 +9,15 @@ import {
9
9
  checkBehavioralCanary,
10
10
  checkCanaries,
11
11
  runBehavioralCanaries
12
- } from "./chunk-NACAGYSY.js";
12
+ } from "./chunk-M4YBQKIJ.js";
13
13
  import {
14
- mintRolloutRows,
15
- rolloutReward
16
- } from "./chunk-IHQDPH7D.js";
14
+ ATIF_SCHEMA_VERSION,
15
+ HARBOR_IMPORT_GAP,
16
+ fromHarborTrajectory,
17
+ relabelImportedSplit,
18
+ toHarborTrajectories,
19
+ toHarborTrajectory
20
+ } from "./chunk-RXHCETDZ.js";
17
21
  import {
18
22
  SUPERVISOR_RUN_SCHEMA,
19
23
  analyzeSupervisorRun,
@@ -27,25 +31,14 @@ import {
27
31
  showMeasured,
28
32
  supervisorRunRolloutLines,
29
33
  writeSupervisorRunReport
30
- } from "./chunk-TSN7JT6D.js";
31
- import "./chunk-VBQ3CRKH.js";
32
- import {
33
- toJsonl,
34
- toRewardRows,
35
- toSftRows
36
- } from "./chunk-EJGRPCO3.js";
37
- import {
38
- ROLLOUT_SCHEMA,
39
- assertRolloutLine,
40
- isRolloutLine,
41
- isTrainableSplit,
42
- validateRolloutLine
43
- } from "./chunk-UWZZKKU7.js";
34
+ } from "./chunk-X4YIBDER.js";
35
+ import "./chunk-HPWUNB47.js";
36
+ import "./chunk-3OCR4R5I.js";
44
37
  import {
45
38
  BENCHMARK_SPLIT_SEED,
46
39
  benchmarks_exports,
47
40
  deterministicSplit
48
- } from "./chunk-XPRT64IE.js";
41
+ } from "./chunk-IYCLP2N2.js";
49
42
  import {
50
43
  DEFAULT_RULES,
51
44
  classifyFailure,
@@ -53,10 +46,7 @@ import {
53
46
  computeToolUseMetrics,
54
47
  iqr,
55
48
  welchsTTest
56
- } from "./chunk-P5W7RQKK.js";
57
- import {
58
- buildTrajectory
59
- } from "./chunk-RZTMDUO7.js";
49
+ } from "./chunk-FXTVJPYD.js";
60
50
  import {
61
51
  analyzeSeries
62
52
  } from "./chunk-BOD4O7OF.js";
@@ -84,7 +74,7 @@ import {
84
74
  harnessAxisOf,
85
75
  parseCorrectnessResponse,
86
76
  verifyCompletion
87
- } from "./chunk-UB2LOJ6Q.js";
77
+ } from "./chunk-QB6BDBP2.js";
88
78
  import {
89
79
  DEFAULT_MUTATION_PRIMITIVES,
90
80
  DEFAULT_RED_TEAM_CORPUS,
@@ -93,20 +83,12 @@ import {
93
83
  JudgeParseError,
94
84
  REFERENCE_EQUIVALENCE_INPUT_LIMITS,
95
85
  REFERENCE_EQUIVALENCE_JUDGE_VERSION,
96
- adversarialJudge,
97
86
  buildReflectionPrompt,
98
- codeExecutionJudge,
99
- coherenceJudge,
100
- costReceiptFromTCloud,
101
- createCustomJudge,
102
- createDomainExpertJudge,
103
87
  createReferenceEquivalenceJudge,
104
88
  crowdingDistance,
105
- defaultJudges,
106
89
  dominates,
107
90
  hashScenarios,
108
91
  llmJudge,
109
- maximumChargeForTCloudRequest,
110
92
  paretoFrontier,
111
93
  paretoFrontierWithCrowding,
112
94
  parseReflectionResponse,
@@ -118,7 +100,7 @@ import {
118
100
  scoreRedTeamOutput,
119
101
  surfaceContentHash,
120
102
  toolNamesForRun
121
- } from "./chunk-NKAGIDE2.js";
103
+ } from "./chunk-2QU3YOPR.js";
122
104
  import {
123
105
  BackendIntegrityError,
124
106
  assertRealAgentReceipts,
@@ -130,7 +112,7 @@ import {
130
112
  inMemoryVerdictCache,
131
113
  summarizeAgentReceiptIntegrity,
132
114
  summarizeBackendIntegrity
133
- } from "./chunk-EZJEIH2R.js";
115
+ } from "./chunk-C6LXANRU.js";
134
116
  import {
135
117
  DEFAULT_COMPLEXITY_WEIGHTS,
136
118
  FindingsStore,
@@ -143,7 +125,7 @@ import {
143
125
  defaultIsMaterial,
144
126
  diffFindings,
145
127
  runSemanticConceptJudge
146
- } from "./chunk-ZUUWPZCV.js";
128
+ } from "./chunk-DODXQREJ.js";
147
129
  import {
148
130
  AnalystRegistry,
149
131
  DEFAULT_TRACE_ANALYST_KINDS,
@@ -160,7 +142,7 @@ import {
160
142
  makeFinding,
161
143
  renderPriorFindings,
162
144
  renderUpstreamFindings
163
- } from "./chunk-DJKY2TSY.js";
145
+ } from "./chunk-BSO5JDQH.js";
164
146
  import "./chunk-HHWE3POT.js";
165
147
  import {
166
148
  DEFAULT_RUN_SCORE_WEIGHTS,
@@ -189,19 +171,39 @@ import {
189
171
  stopOnNoProgress,
190
172
  stopOnRepeatedAction,
191
173
  subjectiveEval
192
- } from "./chunk-BYT7ELPS.js";
174
+ } from "./chunk-WVATSFCP.js";
193
175
  import {
194
176
  assertReleaseConfidence,
195
177
  bootstrapCi,
196
178
  evaluateReleaseConfidence,
197
179
  judgeReplayGate,
198
180
  renderReleaseReport
199
- } from "./chunk-XDWDC2MP.js";
181
+ } from "./chunk-JQSF5DQT.js";
200
182
  import {
201
183
  runEvalCampaign
202
- } from "./chunk-TBL77AUT.js";
203
- import "./chunk-NYLOYM6N.js";
204
- import "./chunk-2MKQIFS4.js";
184
+ } from "./chunk-YQN4ICPP.js";
185
+ import {
186
+ mintRolloutRows
187
+ } from "./chunk-H23X7XKK.js";
188
+ import {
189
+ toJsonl,
190
+ toRewardRows,
191
+ toSftRows
192
+ } from "./chunk-OWN5NPMC.js";
193
+ import {
194
+ ROLLOUT_SCHEMA,
195
+ assertMinted,
196
+ assertMintedLines,
197
+ assertRolloutLine,
198
+ isRolloutLine,
199
+ isTrainableSplit,
200
+ validateRolloutLine
201
+ } from "./chunk-PC5DOSM7.js";
202
+ import {
203
+ buildTrajectory
204
+ } from "./chunk-RZTMDUO7.js";
205
+ import "./chunk-EG66UGL4.js";
206
+ import "./chunk-E7QXT7SX.js";
205
207
  import {
206
208
  LlmCallError,
207
209
  LlmClient,
@@ -217,7 +219,7 @@ import {
217
219
  maximumChargeForLlmRequest,
218
220
  probeLlm,
219
221
  stripFencedJson
220
- } from "./chunk-PBE2LOSS.js";
222
+ } from "./chunk-SFLLL76A.js";
221
223
  import {
222
224
  evaluateInterimReleaseConfidence,
223
225
  pairedEvalueSequence
@@ -228,12 +230,12 @@ import {
228
230
  paretoChart,
229
231
  researchReport,
230
232
  summaryTable
231
- } from "./chunk-VLOATJQ2.js";
233
+ } from "./chunk-TJVT4QFF.js";
232
234
  import {
233
235
  comparePairedArms,
234
236
  pairArms,
235
237
  pairRunRecords
236
- } from "./chunk-DPUHNQLN.js";
238
+ } from "./chunk-7FO3TNPI.js";
237
239
  import {
238
240
  benjaminiHochberg,
239
241
  bonferroni,
@@ -275,7 +277,7 @@ import {
275
277
  weightedMean,
276
278
  wilcoxonSignedRank,
277
279
  wilson
278
- } from "./chunk-MHELPNRP.js";
280
+ } from "./chunk-ZHTZ4EYI.js";
279
281
  import {
280
282
  CostAccountingIncompleteError,
281
283
  CostCallConflictError,
@@ -287,7 +289,7 @@ import {
287
289
  costForTokenPricing,
288
290
  costForUsage,
289
291
  modelPriceKey
290
- } from "./chunk-WS3NZZQQ.js";
292
+ } from "./chunk-VCZ5FQYW.js";
291
293
  import {
292
294
  MODEL_PRICING,
293
295
  MetricsCollector,
@@ -298,8 +300,6 @@ import {
298
300
  resolveModelPricing
299
301
  } from "./chunk-VI2UW6B6.js";
300
302
  import {
301
- FileSystemTraceStore,
302
- InMemoryTraceStore,
303
303
  OTEL_AGENT_EVAL_SCOPE,
304
304
  ReplayCache,
305
305
  ReplayCacheMissError,
@@ -326,7 +326,7 @@ import {
326
326
  scoreTraceInsightReadiness,
327
327
  tokenizeDomainWords,
328
328
  traceAnalystOnRunComplete
329
- } from "./chunk-EOSZT7PL.js";
329
+ } from "./chunk-U4L7JRPZ.js";
330
330
  import "./chunk-7ZZMD7UK.js";
331
331
  import {
332
332
  extractUsage,
@@ -377,10 +377,12 @@ import {
377
377
  traceSpanKindToOpenInferenceKind
378
378
  } from "./chunk-P6FYH6K4.js";
379
379
  import {
380
+ FileSystemTraceStore,
381
+ InMemoryTraceStore,
380
382
  RunIntegrityError,
381
383
  assertRunCaptured,
382
384
  throwIfRunIncomplete
383
- } from "./chunk-TT4KNT67.js";
385
+ } from "./chunk-U4PHLT2N.js";
384
386
  import {
385
387
  FileSystemRawProviderSink,
386
388
  InMemoryRawProviderSink,
@@ -417,7 +419,7 @@ import {
417
419
  validateRunRecord,
418
420
  verifyAgentProfileCell,
419
421
  verifyManifest
420
- } from "./chunk-2JX3CFMB.js";
422
+ } from "./chunk-56TAVBOK.js";
421
423
  import {
422
424
  FAILURE_CLASSES,
423
425
  TRACE_SCHEMA_VERSION,
@@ -427,6 +429,14 @@ import {
427
429
  isSandboxSpan,
428
430
  isToolSpan
429
431
  } from "./chunk-MA6HLL3S.js";
432
+ import {
433
+ isRealnessGated,
434
+ observedScore,
435
+ observedSplitScore,
436
+ scoreOrigin,
437
+ trainingReward,
438
+ trainingScore
439
+ } from "./chunk-OIUOT4QD.js";
430
440
  import {
431
441
  AgentEvalError,
432
442
  CaptureIntegrityError,
@@ -764,15 +774,11 @@ function renderCommitMessage(input) {
764
774
  }
765
775
 
766
776
  // src/executor.ts
767
- function describeShape(content, message) {
768
- if (message === void 0 || message === null) return "no choices[0].message";
769
- return `message.content of type ${content === null ? "null" : typeof content}`;
770
- }
771
777
  function errMessage(err) {
772
778
  if (err instanceof Error) return `${err.name}: ${err.message}`;
773
779
  return String(err);
774
780
  }
775
- async function executeScenario(tc, scenario, config) {
781
+ async function executeScenario(chat, scenario, config) {
776
782
  const startTime = Date.now();
777
783
  const model = config.model ?? "gpt-4o";
778
784
  const costLedger = config.costLedger ?? new CostLedger();
@@ -803,19 +809,18 @@ async function executeScenario(tc, scenario, config) {
803
809
  phase: config.costPhase ?? "benchmark.agent",
804
810
  actor: "scenario-agent",
805
811
  model,
806
- maximumCharge: maximumChargeForTCloudRequest(request, config.tcloudMaximumAttempts),
812
+ maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
807
813
  tags: costTags,
808
814
  signal: config.signal,
809
- execute: () => tc.chat(request),
810
- receipt: (response) => costReceiptFromTCloud(response, model)
815
+ execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
816
+ receipt: costReceiptFromLlm,
817
+ receiptFromError: costReceiptFromLlmError
811
818
  });
812
819
  if (!paid.succeeded) throw paid.error;
813
- const resp = paid.value;
814
- const message = resp.choices?.[0]?.message;
815
- const rawContent = message?.content;
816
- if (message === void 0 || message === null || typeof rawContent !== "string") {
820
+ const rawContent = paid.value.content;
821
+ if (typeof rawContent !== "string") {
817
822
  throw new CaptureIntegrityError(
818
- `chat response for scenario "${scenario.id}" turn ${i} is malformed: expected choices[0].message.content to be a string, got ${describeShape(rawContent, message)}`
823
+ `chat response for scenario "${scenario.id}" turn ${i} is malformed: expected content to be a string, got ${rawContent === null ? "null" : typeof rawContent}`
819
824
  );
820
825
  }
821
826
  const content = rawContent;
@@ -903,8 +908,7 @@ async function executeScenario(tc, scenario, config) {
903
908
  costLedger,
904
909
  costPhase: config.costPhase ?? "benchmark.judge",
905
910
  costTags,
906
- signal: config.signal,
907
- tcloudMaximumAttempts: config.tcloudMaximumAttempts
911
+ signal: config.signal
908
912
  };
909
913
  const judgeResults = [];
910
914
  let failedJudges = 0;
@@ -923,7 +927,7 @@ async function executeScenario(tc, scenario, config) {
923
927
  console.log(` ${judgeName} retry ${attempt}/2 (waiting ${wait / 1e3}s)`);
924
928
  await sleep2(wait);
925
929
  }
926
- const scores = await judge(tc, judgeInput);
930
+ const scores = await judge(chat, judgeInput);
927
931
  judgeResults.push(scores);
928
932
  await sleep2(3e3);
929
933
  lastError = void 0;
@@ -983,10 +987,10 @@ async function executeScenario(tc, scenario, config) {
983
987
 
984
988
  // src/benchmark.ts
985
989
  var BenchmarkRunner = class {
986
- tc;
990
+ chat;
987
991
  config;
988
- constructor(tc, config) {
989
- this.tc = tc;
992
+ constructor(chat, config) {
993
+ this.chat = chat;
990
994
  this.config = config;
991
995
  }
992
996
  async run(scenarios) {
@@ -1008,13 +1012,12 @@ var BenchmarkRunner = class {
1008
1012
  console.log(`[${i + 1}/${toRun.length}] ${scenario.id} (${scenario.persona})`);
1009
1013
  console.log(` thesis: ${scenario.thesis}`);
1010
1014
  console.log(` turns: ${scenario.turns.length}`);
1011
- const result = await executeScenario(this.tc, scenario, {
1015
+ const result = await executeScenario(this.chat, scenario, {
1012
1016
  systemPrompt: this.config.systemPrompt,
1013
1017
  model: this.config.model,
1014
1018
  judges: this.config.judges,
1015
1019
  costLedger,
1016
- costTags,
1017
- tcloudMaximumAttempts: this.config.tcloudMaximumAttempts
1020
+ costTags
1018
1021
  });
1019
1022
  results.push(result);
1020
1023
  for (const turn of result.turns) {
@@ -1717,19 +1720,17 @@ var RIGOR_STANCE = {
1717
1720
  relentless: "Your stance: a senior partner reviewing this work for a client who will litigate if it is wrong. You interrogate every claim. You accept nothing undefended. You find the single weakest point in every answer and attack it. Courteous, never satisfied."
1718
1721
  };
1719
1722
  var AgentDriver = class {
1720
- tc;
1723
+ chat;
1721
1724
  client;
1722
1725
  driverModel;
1723
1726
  productContext;
1724
1727
  costLedger;
1725
- tcloudMaximumAttempts;
1726
- constructor(tc, config) {
1727
- this.tc = tc;
1728
+ constructor(chat, config) {
1729
+ this.chat = chat;
1728
1730
  this.client = config.client;
1729
1731
  this.driverModel = config.driverModel ?? "claude-sonnet-4-6";
1730
1732
  this.productContext = config.productContext ?? "";
1731
1733
  this.costLedger = config.costLedger ?? new CostLedger();
1732
- this.tcloudMaximumAttempts = config.tcloudMaximumAttempts;
1733
1734
  }
1734
1735
  /**
1735
1736
  * Run a persona through the product.
@@ -1812,15 +1813,14 @@ var AgentDriver = class {
1812
1813
  }
1813
1814
  /** Use the driver LLM to decide what the "user" says next */
1814
1815
  async decideNextMessage(persona, state, history, costTags) {
1815
- return decideNextUserTurn(this.tc, {
1816
+ return decideNextUserTurn(this.chat, {
1816
1817
  persona,
1817
1818
  state,
1818
1819
  history,
1819
1820
  productContext: this.productContext,
1820
1821
  model: this.driverModel,
1821
1822
  costLedger: this.costLedger,
1822
- costTags,
1823
- tcloudMaximumAttempts: this.tcloudMaximumAttempts
1823
+ costTags
1824
1824
  });
1825
1825
  }
1826
1826
  /** Handle pending approvals based on persona feedback patterns */
@@ -1925,7 +1925,7 @@ COMPLETION: drive toward the goal's real acceptance check. Do not declare done \
1925
1925
 
1926
1926
  Output ONLY your next instruction to the worker \u2014 direct, detailed, actionable, in the first person as the driver. No meta-commentary, no preamble.`;
1927
1927
  }
1928
- async function decideNextUserTurn(tc, opts) {
1928
+ async function decideNextUserTurn(chat, opts) {
1929
1929
  const { persona, state, history, productContext = "", model = "claude-sonnet-4-6" } = opts;
1930
1930
  const lastResponse = history.length > 0 ? history[history.length - 1].content.slice(0, 2e3) : "(no conversation yet \u2014 this is the first message)";
1931
1931
  const recentHistory = history.slice(-6).map((h) => `${h.role}: ${h.content.slice(0, 500)}`).join("\n\n");
@@ -1951,14 +1951,13 @@ ${lastResponse}` : "No conversation yet. Send your opening message \u2014 in cha
1951
1951
  actor: "decideNextUserTurn",
1952
1952
  model,
1953
1953
  tags: opts.costTags,
1954
- maximumCharge: maximumChargeForTCloudRequest(request, opts.tcloudMaximumAttempts),
1955
- execute: () => tc.chat(request),
1956
- receipt: (response) => costReceiptFromTCloud(response, model)
1954
+ maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
1955
+ execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
1956
+ receipt: costReceiptFromLlm,
1957
+ receiptFromError: costReceiptFromLlmError
1957
1958
  });
1958
1959
  if (!paid.succeeded) throw paid.error;
1959
- const resp = paid.value;
1960
- const content = resp.choices?.[0]?.message?.content ?? "";
1961
- return content.trim();
1960
+ return paid.value.content.trim();
1962
1961
  }
1963
1962
 
1964
1963
  // src/feedback-trajectory.ts
@@ -4842,7 +4841,7 @@ function assertNonNegative(n, name) {
4842
4841
 
4843
4842
  // src/eval-trace-store.ts
4844
4843
  function runScore(record) {
4845
- return runTaskScore(record);
4844
+ return trainingScore(record);
4846
4845
  }
4847
4846
  function matches(record, f) {
4848
4847
  if (f.experimentId && record.experimentId !== f.experimentId) return false;
@@ -4938,13 +4937,19 @@ var EvalTraceStore = class {
4938
4937
  * Highest-scoring run for a scenario (optionally restricted to a candidate).
4939
4938
  * Returns null when no run matches. Ties resolve to the earliest-appended run
4940
4939
  * so the result is stable.
4940
+ *
4941
+ * Runs flagged as gamed are DROPPED, not zeroed. The caller's use for this is
4942
+ * few-shot seeding — the returned trajectory becomes an example to copy — so
4943
+ * the same rule as SFT applies: a faked success must not be in the candidate
4944
+ * set at all. When every run for the scenario is gated the honest answer is
4945
+ * `null` (no exemplar), never the least-bad fake.
4941
4946
  */
4942
4947
  async getBest(scenarioId, opts = {}) {
4943
- const rows = await this.query({
4948
+ const rows = (await this.query({
4944
4949
  scenarioId,
4945
4950
  candidateId: opts.candidateId,
4946
4951
  splitTag: opts.splitTag
4947
- });
4952
+ })).filter((r) => !isRealnessGated(r));
4948
4953
  const scored = rows.flatMap((record) => {
4949
4954
  const score = runScore(record);
4950
4955
  return score === void 0 ? [] : [{ record, score }];
@@ -4966,6 +4971,9 @@ var EvalTraceStore = class {
4966
4971
  * ran a scenario more than once, its best `runScore` for that scenario is
4967
4972
  * used. Throws when there is no paired scenario — an unpaired "comparison" is
4968
4973
  * not one.
4974
+ *
4975
+ * Realness-gated runs are excluded and counted in `realnessGatedRuns`, never
4976
+ * folded in as a zero.
4969
4977
  */
4970
4978
  async compareRuns(candidateA, candidateB) {
4971
4979
  if (candidateA === candidateB) {
@@ -4974,12 +4982,17 @@ var EvalTraceStore = class {
4974
4982
  );
4975
4983
  }
4976
4984
  const rows = await this.backend.load();
4985
+ let realnessGatedRuns = 0;
4977
4986
  const bestByScenario = (candidate) => {
4978
4987
  const m = /* @__PURE__ */ new Map();
4979
4988
  for (const r of rows) {
4980
4989
  if (r.candidateId !== candidate) continue;
4981
4990
  const sid = r.scenarioId;
4982
4991
  if (!sid) continue;
4992
+ if (isRealnessGated(r)) {
4993
+ realnessGatedRuns++;
4994
+ continue;
4995
+ }
4983
4996
  const s = runScore(r);
4984
4997
  if (s === void 0) continue;
4985
4998
  const prev = m.get(sid);
@@ -4992,7 +5005,7 @@ var EvalTraceStore = class {
4992
5005
  const paired = [...aScores.keys()].filter((sid) => bScores.has(sid)).sort();
4993
5006
  if (paired.length === 0) {
4994
5007
  throw new ValidationError(
4995
- `EvalTraceStore.compareRuns: "${candidateA}" and "${candidateB}" share no scenario (need scenarioId on records)`
5008
+ realnessGatedRuns > 0 ? `EvalTraceStore.compareRuns: "${candidateA}" and "${candidateB}" share no scenario with honest runs on both sides (${realnessGatedRuns} run(s) were realness-gated)` : `EvalTraceStore.compareRuns: "${candidateA}" and "${candidateB}" share no scenario (need scenarioId on records)`
4996
5009
  );
4997
5010
  }
4998
5011
  let sumA = 0;
@@ -5020,7 +5033,8 @@ var EvalTraceStore = class {
5020
5033
  meanDelta: meanB - meanA,
5021
5034
  bWins,
5022
5035
  ties,
5023
- aWins
5036
+ aWins,
5037
+ realnessGatedRuns
5024
5038
  };
5025
5039
  }
5026
5040
  };
@@ -9610,16 +9624,20 @@ var HeldOutGate = class {
9610
9624
  const candidateId = inferCandidateId2(candidate, this.baselineKey);
9611
9625
  const baselineId = this.baselineKey;
9612
9626
  assertScenarioIdentities([...candidate, ...baseline]);
9613
- const candidateSearch = scoredRuns(candidate, "searchScore", "search");
9614
- const baselineSearch = scoredRuns(baseline, "searchScore", "search");
9615
- const candidateHoldout = scoredRuns(candidate, "holdoutScore", "holdout");
9616
- const baselineHoldout = scoredRuns(baseline, "holdoutScore", "holdout");
9627
+ const realnessGatedRuns = [...candidate, ...baseline].filter(isRealnessGated).length;
9628
+ const honestCandidate = candidate.filter((run) => !isRealnessGated(run));
9629
+ const honestBaseline = baseline.filter((run) => !isRealnessGated(run));
9630
+ const candidateSearch = scoredRuns(honestCandidate, "search");
9631
+ const baselineSearch = scoredRuns(honestBaseline, "search");
9632
+ const candidateHoldout = scoredRuns(honestCandidate, "holdout");
9633
+ const baselineHoldout = scoredRuns(honestBaseline, "holdout");
9617
9634
  const searchPairing = pairRunRecords(baselineSearch, candidateSearch);
9618
9635
  const holdoutPairing = pairRunRecords(baselineHoldout, candidateHoldout);
9619
- const beforeSearch = searchPairing.pairs.map((pair) => pair.baseline.outcome.searchScore);
9620
- const afterSearch = searchPairing.pairs.map((pair) => pair.treatment.outcome.searchScore);
9621
- const beforeHoldout = holdoutPairing.pairs.map((pair) => pair.baseline.outcome.holdoutScore);
9622
- const afterHoldout = holdoutPairing.pairs.map((pair) => pair.treatment.outcome.holdoutScore);
9636
+ const splitScoreOf = (run, split) => observedSplitScore(run, split);
9637
+ const beforeSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "search"));
9638
+ const afterSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "search"));
9639
+ const beforeHoldout = holdoutPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "holdout"));
9640
+ const afterHoldout = holdoutPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "holdout"));
9623
9641
  const productiveRuns = beforeHoldout.length;
9624
9642
  const candidateSearchMean = meanOrNull(afterSearch);
9625
9643
  const candidateHoldoutMean = meanOrNull(afterHoldout);
@@ -9638,7 +9656,8 @@ var HeldOutGate = class {
9638
9656
  overfitGap,
9639
9657
  baselineOverfitGap,
9640
9658
  medianCandidateCost,
9641
- medianBaselineCost
9659
+ medianBaselineCost,
9660
+ realnessGatedRuns
9642
9661
  };
9643
9662
  const missingSplitScores = [
9644
9663
  candidateSearch.length === 0 ? "candidate search" : null,
@@ -9755,10 +9774,12 @@ function assertScenarioIdentities(runs) {
9755
9774
  }
9756
9775
  }
9757
9776
  }
9758
- function scoredRuns(runs, field, splitFilter) {
9759
- return runs.filter(
9760
- (run) => run.splitTag === splitFilter && typeof run.outcome[field] === "number" && Number.isFinite(run.outcome[field])
9761
- );
9777
+ function scoredRuns(runs, split) {
9778
+ return runs.filter((run) => {
9779
+ if (run.splitTag !== split) return false;
9780
+ const v = observedSplitScore(run, split);
9781
+ return typeof v === "number" && Number.isFinite(v);
9782
+ });
9762
9783
  }
9763
9784
  function meanOrNull(xs) {
9764
9785
  if (xs.length === 0) return null;
@@ -10315,7 +10336,7 @@ async function tracedAnalyzeTraces(input, options, traceOpts) {
10315
10336
 
10316
10337
  // src/traced-judges.ts
10317
10338
  function traceJudge(judge, judgeName, opts) {
10318
- return async (tc, input) => {
10339
+ return async (chat, input) => {
10319
10340
  const span = await opts.emitter.span({
10320
10341
  kind: "llm",
10321
10342
  name: `judge:${judgeName}`,
@@ -10326,7 +10347,7 @@ function traceJudge(judge, judgeName, opts) {
10326
10347
  }
10327
10348
  });
10328
10349
  try {
10329
- const scores = await judge(tc, input);
10350
+ const scores = await judge(chat, input);
10330
10351
  const composite = scores.length > 0 ? scores.reduce((sum4, s) => sum4 + s.score, 0) / scores.length : 0;
10331
10352
  await span.end({
10332
10353
  attributes: {
@@ -10344,7 +10365,7 @@ function traceJudge(judge, judgeName, opts) {
10344
10365
  };
10345
10366
  }
10346
10367
  function traceJudgeEnsemble(judges, judgeNames, opts) {
10347
- return async (tc, input) => {
10368
+ return async (chat, input) => {
10348
10369
  const ensembleSpan = await opts.emitter.span({
10349
10370
  kind: "custom",
10350
10371
  name: "judge:ensemble",
@@ -10365,7 +10386,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
10365
10386
  parentSpanId: ensembleSpan.span.spanId
10366
10387
  });
10367
10388
  try {
10368
- const scores = await tracedFn(tc, input);
10389
+ const scores = await tracedFn(chat, input);
10369
10390
  allScores.push(...scores);
10370
10391
  } catch (err) {
10371
10392
  if (!(err instanceof JudgeParseError)) throw err;
@@ -11387,6 +11408,7 @@ ${failures.join("\n")}`);
11387
11408
  }
11388
11409
  export {
11389
11410
  AGENT_PROFILE_KINDS,
11411
+ ATIF_SCHEMA_VERSION,
11390
11412
  ATTESTATION_ALGORITHM,
11391
11413
  AgentDriver,
11392
11414
  AgentEvalError,
@@ -11439,6 +11461,7 @@ export {
11439
11461
  FileSystemRawProviderSink,
11440
11462
  FileSystemTraceStore,
11441
11463
  FindingsStore,
11464
+ HARBOR_IMPORT_GAP,
11442
11465
  HARNESS_NATIVE_MODEL,
11443
11466
  HeldOutGate,
11444
11467
  HoldoutAuditor,
@@ -11532,7 +11555,6 @@ export {
11532
11555
  ValidationError,
11533
11556
  VerificationError,
11534
11557
  acquisitionPlansForKnowledgeGaps,
11535
- adversarialJudge,
11536
11558
  agentProfileCellHashMaterial,
11537
11559
  agentProfileCellKey,
11538
11560
  agentProfileHash,
@@ -11558,6 +11580,8 @@ export {
11558
11580
  assertCapabilityHeadroom,
11559
11581
  assertCrossFamily,
11560
11582
  assertLlmRoute,
11583
+ assertMinted,
11584
+ assertMintedLines,
11561
11585
  assertModelsServed,
11562
11586
  assertNoHiddenLeak,
11563
11587
  assertProductBenchmarkRun,
@@ -11617,9 +11641,7 @@ export {
11617
11641
  claudeCodeSupervisorRunReader,
11618
11642
  cliffsDelta,
11619
11643
  clusteredPairedBinary,
11620
- codeExecutionJudge,
11621
11644
  cohensD,
11622
- coherenceJudge,
11623
11645
  collectionPreserved,
11624
11646
  commentsForSource,
11625
11647
  commitBisect,
@@ -11654,9 +11676,7 @@ export {
11654
11676
  createAnalystAi,
11655
11677
  createAntiSlopJudge,
11656
11678
  createChatClient,
11657
- createCustomJudge,
11658
11679
  createDefaultReviewer,
11659
- createDomainExpertJudge,
11660
11680
  createFeedbackTrajectory,
11661
11681
  createIntentMatchJudge,
11662
11682
  createLlmCorrectnessChecker,
@@ -11677,7 +11697,6 @@ export {
11677
11697
  decideReferenceReplayRunPromotion,
11678
11698
  defaultBlendWeights,
11679
11699
  defaultIsMaterial,
11680
- defaultJudges,
11681
11700
  defaultProviderRedactor,
11682
11701
  defaultReferenceReplayMatcher,
11683
11702
  defaultTraceInsightPanel,
@@ -11738,6 +11757,7 @@ export {
11738
11757
  formatDriverReport,
11739
11758
  formatFindings,
11740
11759
  formatScorecardDiff,
11760
+ fromHarborTrajectory,
11741
11761
  gainHistogram,
11742
11762
  gateTreatmentApplied,
11743
11763
  gateTreatmentFromMetrics,
@@ -11777,6 +11797,7 @@ export {
11777
11797
  isModelPriced,
11778
11798
  isOtelConfigured,
11779
11799
  isOtlpModelCall,
11800
+ isRealnessGated,
11780
11801
  isRetrievalSpan,
11781
11802
  isRolloutLine,
11782
11803
  isRunRecord,
@@ -11829,6 +11850,8 @@ export {
11829
11850
  notBlocked,
11830
11851
  objectiveEval,
11831
11852
  observeAll,
11853
+ observedScore,
11854
+ observedSplitScore,
11832
11855
  otelRunCompleteHook,
11833
11856
  otlpRowsToRunRecords,
11834
11857
  otlpRowsToTraceRunRecords,
@@ -11891,6 +11914,7 @@ export {
11891
11914
  referenceReplayScenarioToRunScore,
11892
11915
  regexMatch,
11893
11916
  regexMatches,
11917
+ relabelImportedSplit,
11894
11918
  renderMarkdownReport,
11895
11919
  renderPlaybookMarkdown,
11896
11920
  renderPreferenceMemoryMarkdown,
@@ -11911,7 +11935,6 @@ export {
11911
11935
  researchReport,
11912
11936
  resolveModelPricing,
11913
11937
  resolveSeat,
11914
- rolloutReward,
11915
11938
  rollupSupervisorRuns,
11916
11939
  roundTripRunRecord,
11917
11940
  routeFields,
@@ -11948,6 +11971,7 @@ export {
11948
11971
  scoreContinuity,
11949
11972
  scoreFromEvals,
11950
11973
  scoreKnowledgeReadiness,
11974
+ scoreOrigin,
11951
11975
  scorePrReviewComments,
11952
11976
  scorePrReviewSource,
11953
11977
  scoreRedTeamOutput,
@@ -11979,6 +12003,8 @@ export {
11979
12003
  textInSnapshot,
11980
12004
  throwIfRunIncomplete,
11981
12005
  toAgentProfileJson,
12006
+ toHarborTrajectories,
12007
+ toHarborTrajectory,
11982
12008
  toJsonl,
11983
12009
  toLangfuseEnvelope,
11984
12010
  toOpenAiTool,
@@ -11995,6 +12021,8 @@ export {
11995
12021
  traceJudgeEnsemble,
11996
12022
  traceSpanKindToOpenInferenceKind,
11997
12023
  tracedAnalyzeTraces,
12024
+ trainingReward,
12025
+ trainingScore,
11998
12026
  typoMutator,
11999
12027
  urlContains,
12000
12028
  userQuestionsForKnowledgeGaps,