@tangle-network/agent-eval 0.124.0 → 0.126.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/CHANGELOG.md +60 -35
  2. package/README.md +270 -189
  3. package/dist/analyst/index.d.ts +15 -145
  4. package/dist/analyst/index.js +33 -47
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/benchmarks/index.d.ts +45 -162
  7. package/dist/benchmarks/index.js +8 -9
  8. package/dist/campaign/index.d.ts +3655 -5365
  9. package/dist/campaign/index.js +21 -95
  10. package/dist/{chunk-R226UZOI.js → chunk-474LBSOX.js} +2 -2
  11. package/dist/{chunk-W5B3ZGP3.js → chunk-4B7ZZHPX.js} +8 -6
  12. package/dist/{chunk-W5B3ZGP3.js.map → chunk-4B7ZZHPX.js.map} +1 -1
  13. package/dist/{chunk-DT7OXY3C.js → chunk-CM4OILD2.js} +535 -846
  14. package/dist/chunk-CM4OILD2.js.map +1 -0
  15. package/dist/{chunk-HM6V7F3M.js → chunk-FO7HEH76.js} +3 -3
  16. package/dist/chunk-IILEIWGW.js +635 -0
  17. package/dist/chunk-IILEIWGW.js.map +1 -0
  18. package/dist/{chunk-EQUK3RFS.js → chunk-J5SQWP6Y.js} +8 -5
  19. package/dist/chunk-J5SQWP6Y.js.map +1 -0
  20. package/dist/chunk-KO2PZOGP.js +4637 -0
  21. package/dist/chunk-KO2PZOGP.js.map +1 -0
  22. package/dist/{chunk-4Y7AAATF.js → chunk-LKKT3IVV.js} +574 -81
  23. package/dist/chunk-LKKT3IVV.js.map +1 -0
  24. package/dist/chunk-M7AH34KV.js +155 -0
  25. package/dist/chunk-M7AH34KV.js.map +1 -0
  26. package/dist/chunk-NTOV7RU5.js +7152 -0
  27. package/dist/chunk-NTOV7RU5.js.map +1 -0
  28. package/dist/{chunk-QFQZ3U3X.js → chunk-OCFJACJU.js} +2 -2
  29. package/dist/{chunk-GID26AN4.js → chunk-P22LJ3Y2.js} +4 -6
  30. package/dist/{chunk-GID26AN4.js.map → chunk-P22LJ3Y2.js.map} +1 -1
  31. package/dist/{chunk-SJT4OBVL.js → chunk-SDPM6554.js} +3 -3
  32. package/dist/{chunk-D5JZ7UDZ.js → chunk-UCLVDLCH.js} +136 -50
  33. package/dist/chunk-UCLVDLCH.js.map +1 -0
  34. package/dist/chunk-UI4YMIN2.js +105 -0
  35. package/dist/chunk-UI4YMIN2.js.map +1 -0
  36. package/dist/chunk-VBQ3CRKH.js +291 -0
  37. package/dist/chunk-VBQ3CRKH.js.map +1 -0
  38. package/dist/{chunk-JKDNAOF5.js → chunk-W4L6C2XT.js} +2 -2
  39. package/dist/{chunk-GRCDRKII.js → chunk-WS3NZZQQ.js} +58 -20
  40. package/dist/chunk-WS3NZZQQ.js.map +1 -0
  41. package/dist/cli.js +3 -3
  42. package/dist/contract/index.d.ts +3221 -3094
  43. package/dist/contract/index.js +173 -42
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/control.js +2 -3
  46. package/dist/fuzz.d.ts +14 -1
  47. package/dist/fuzz.js +1 -1
  48. package/dist/hosted/index.d.ts +8 -1
  49. package/dist/index.d.ts +208 -690
  50. package/dist/index.js +185 -500
  51. package/dist/index.js.map +1 -1
  52. package/dist/openapi.json +1 -1
  53. package/dist/rl.d.ts +5 -100
  54. package/dist/rl.js +4 -5
  55. package/dist/rl.js.map +1 -1
  56. package/dist/rollout/index.d.ts +9 -1
  57. package/dist/rollout/index.js +6 -6
  58. package/dist/{run-campaign-I3JXKVAK.js → run-campaign-LVFKZCEU.js} +3 -3
  59. package/dist/supervisor-run/index.d.ts +156 -4
  60. package/dist/supervisor-run/index.js +14 -2
  61. package/dist/traces.js +2 -3
  62. package/dist/wire/index.d.ts +14 -1
  63. package/dist/wire/index.js +3 -3
  64. package/docs/campaign-proposers.md +363 -168
  65. package/docs/design/loop-taxonomy.md +142 -190
  66. package/docs/design.md +1 -1
  67. package/docs/distributed-driver.md +8 -11
  68. package/docs/feature-guide.md +20 -19
  69. package/docs/knowledge-readiness.md +2 -5
  70. package/docs/multi-shot-optimization.md +35 -27
  71. package/docs/rollout.md +5 -5
  72. package/package.json +4 -4
  73. package/dist/chunk-4Y7AAATF.js.map +0 -1
  74. package/dist/chunk-5PVZVCZB.js +0 -9190
  75. package/dist/chunk-5PVZVCZB.js.map +0 -1
  76. package/dist/chunk-A6GT67HT.js +0 -550
  77. package/dist/chunk-A6GT67HT.js.map +0 -1
  78. package/dist/chunk-D5JZ7UDZ.js.map +0 -1
  79. package/dist/chunk-DT7OXY3C.js.map +0 -1
  80. package/dist/chunk-EQUK3RFS.js.map +0 -1
  81. package/dist/chunk-GC4ATIKK.js +0 -317
  82. package/dist/chunk-GC4ATIKK.js.map +0 -1
  83. package/dist/chunk-GRCDRKII.js.map +0 -1
  84. package/dist/chunk-LOW3U7JZ.js +0 -328
  85. package/dist/chunk-LOW3U7JZ.js.map +0 -1
  86. package/dist/chunk-MGGFVCJ7.js +0 -288
  87. package/dist/chunk-MGGFVCJ7.js.map +0 -1
  88. package/dist/chunk-PMITBABE.js +0 -3841
  89. package/dist/chunk-PMITBABE.js.map +0 -1
  90. package/dist/chunk-R7ZRE2KV.js +0 -138
  91. package/dist/chunk-R7ZRE2KV.js.map +0 -1
  92. /package/dist/{chunk-R226UZOI.js.map → chunk-474LBSOX.js.map} +0 -0
  93. /package/dist/{chunk-HM6V7F3M.js.map → chunk-FO7HEH76.js.map} +0 -0
  94. /package/dist/{chunk-QFQZ3U3X.js.map → chunk-OCFJACJU.js.map} +0 -0
  95. /package/dist/{chunk-SJT4OBVL.js.map → chunk-SDPM6554.js.map} +0 -0
  96. /package/dist/{chunk-JKDNAOF5.js.map → chunk-W4L6C2XT.js.map} +0 -0
  97. /package/dist/{run-campaign-I3JXKVAK.js.map → run-campaign-LVFKZCEU.js.map} +0 -0
package/dist/index.js CHANGED
@@ -9,24 +9,26 @@ import {
9
9
  checkBehavioralCanary,
10
10
  checkCanaries,
11
11
  runBehavioralCanaries
12
- } from "./chunk-SJT4OBVL.js";
12
+ } from "./chunk-SDPM6554.js";
13
13
  import {
14
14
  mintRolloutRows,
15
15
  rolloutReward
16
- } from "./chunk-MGGFVCJ7.js";
16
+ } from "./chunk-M7AH34KV.js";
17
17
  import {
18
18
  SUPERVISOR_RUN_SCHEMA,
19
19
  analyzeSupervisorRun,
20
20
  analyzeSupervisorRunSources,
21
+ claudeCodeSupervisorRunReader,
21
22
  isUnavailable,
23
+ readClaudeCodeSupervisorRun,
22
24
  renderSupervisorRunHeadline,
23
25
  renderSupervisorRunMarkdown,
24
26
  rollupSupervisorRuns,
25
27
  showMeasured,
26
28
  supervisorRunRolloutLines,
27
29
  writeSupervisorRunReport
28
- } from "./chunk-4Y7AAATF.js";
29
- import "./chunk-R7ZRE2KV.js";
30
+ } from "./chunk-LKKT3IVV.js";
31
+ import "./chunk-VBQ3CRKH.js";
30
32
  import {
31
33
  toJsonl,
32
34
  toRewardRows,
@@ -44,7 +46,7 @@ import {
44
46
  BENCHMARK_SPLIT_SEED,
45
47
  benchmarks_exports,
46
48
  deterministicSplit
47
- } from "./chunk-JKDNAOF5.js";
49
+ } from "./chunk-W4L6C2XT.js";
48
50
  import {
49
51
  DEFAULT_RULES,
50
52
  classifyFailure,
@@ -85,7 +87,7 @@ import {
85
87
  pairArms,
86
88
  parseCorrectnessResponse,
87
89
  verifyCompletion
88
- } from "./chunk-5PVZVCZB.js";
90
+ } from "./chunk-KO2PZOGP.js";
89
91
  import {
90
92
  DEFAULT_MUTATION_PRIMITIVES,
91
93
  DEFAULT_RED_TEAM_CORPUS,
@@ -96,7 +98,6 @@ import {
96
98
  REFERENCE_EQUIVALENCE_JUDGE_VERSION,
97
99
  adversarialJudge,
98
100
  buildReflectionPrompt,
99
- campaignMeanComposite,
100
101
  codeExecutionJudge,
101
102
  coherenceJudge,
102
103
  costReceiptFromTCloud,
@@ -106,9 +107,7 @@ import {
106
107
  crowdingDistance,
107
108
  defaultJudges,
108
109
  dominates,
109
- gepaProposer,
110
110
  hashScenarios,
111
- heldOutGate,
112
111
  llmJudge,
113
112
  maximumChargeForTCloudRequest,
114
113
  paretoFrontier,
@@ -117,13 +116,12 @@ import {
117
116
  redTeamDataset,
118
117
  redTeamReport,
119
118
  runCanaries,
120
- runImprovementLoop,
121
119
  runReferenceEquivalenceJudge,
122
120
  scalarScore,
123
121
  scoreRedTeamOutput,
124
122
  surfaceContentHash,
125
123
  toolNamesForRun
126
- } from "./chunk-PMITBABE.js";
124
+ } from "./chunk-NTOV7RU5.js";
127
125
  import {
128
126
  BackendIntegrityError,
129
127
  assertRealAgentReceipts,
@@ -135,7 +133,7 @@ import {
135
133
  inMemoryVerdictCache,
136
134
  summarizeAgentReceiptIntegrity,
137
135
  summarizeBackendIntegrity
138
- } from "./chunk-D5JZ7UDZ.js";
136
+ } from "./chunk-UCLVDLCH.js";
139
137
  import {
140
138
  DEFAULT_COMPLEXITY_WEIGHTS,
141
139
  FindingsStore,
@@ -148,47 +146,32 @@ import {
148
146
  defaultIsMaterial,
149
147
  diffFindings,
150
148
  runSemanticConceptJudge
151
- } from "./chunk-W5B3ZGP3.js";
152
- import {
153
- buildDefaultAnalystRegistry,
154
- computeTraceMetrics,
155
- createChatClient
156
- } from "./chunk-A6GT67HT.js";
157
- import "./chunk-HHWE3POT.js";
149
+ } from "./chunk-4B7ZZHPX.js";
158
150
  import {
159
151
  AnalystRegistry,
160
- DEFAULT_RUN_SCORE_WEIGHTS,
161
152
  DEFAULT_TRACE_ANALYST_KINDS,
162
153
  FAILURE_MODE_KIND_SPEC,
163
154
  IMPROVEMENT_KIND_SPEC,
164
155
  KNOWLEDGE_GAP_KIND_SPEC,
165
156
  KNOWLEDGE_POISONING_KIND_SPEC,
166
- Mutex,
167
- POLICY_EDIT_AXES,
168
- POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
169
- POLICY_EDIT_TARGET_SURFACES,
170
- PolicyEditValidationError,
171
- admitPolicyEdit,
172
- aggregateRunScore,
173
- applyPolicyEditToSurface,
174
- clamp01,
157
+ buildDefaultAnalystRegistry,
175
158
  computeFindingId,
176
- computePolicyEditId,
159
+ computeTraceMetrics,
177
160
  createAnalystAi,
161
+ createChatClient,
178
162
  createTraceAnalystKind,
179
- isPolicyEdit,
180
163
  makeFinding,
181
- makePolicyEdit,
182
- makePolicyEditCandidateRecord,
183
- mapConcurrent,
184
- policyEditFromFinding,
185
- policyEditsFromFindings,
186
164
  renderPriorFindings,
187
- renderUpstreamFindings,
188
- scorePolicyEditReadiness,
189
- validatePolicyEdit,
190
- validatePolicyEditCandidateRecord
191
- } from "./chunk-DT7OXY3C.js";
165
+ renderUpstreamFindings
166
+ } from "./chunk-CM4OILD2.js";
167
+ import "./chunk-HHWE3POT.js";
168
+ import {
169
+ DEFAULT_RUN_SCORE_WEIGHTS,
170
+ Mutex,
171
+ aggregateRunScore,
172
+ clamp01,
173
+ mapConcurrent
174
+ } from "./chunk-UI4YMIN2.js";
192
175
  import {
193
176
  allCriticalPassed,
194
177
  controlFailureClassFromVerification,
@@ -209,7 +192,7 @@ import {
209
192
  stopOnNoProgress,
210
193
  stopOnRepeatedAction,
211
194
  subjectiveEval
212
- } from "./chunk-R226UZOI.js";
195
+ } from "./chunk-474LBSOX.js";
213
196
  import {
214
197
  assertReleaseConfidence,
215
198
  bootstrapCi,
@@ -219,7 +202,7 @@ import {
219
202
  } from "./chunk-MOXWMGPC.js";
220
203
  import {
221
204
  runEvalCampaign
222
- } from "./chunk-GID26AN4.js";
205
+ } from "./chunk-P22LJ3Y2.js";
223
206
  import "./chunk-ARU2PZFM.js";
224
207
  import {
225
208
  LlmCallError,
@@ -236,7 +219,7 @@ import {
236
219
  maximumChargeForLlmRequest,
237
220
  probeLlm,
238
221
  stripFencedJson
239
- } from "./chunk-EQUK3RFS.js";
222
+ } from "./chunk-J5SQWP6Y.js";
240
223
  import {
241
224
  evaluateInterimReleaseConfidence,
242
225
  pairedEvalueSequence
@@ -299,7 +282,7 @@ import {
299
282
  costForTokenPricing,
300
283
  costForUsage,
301
284
  modelPriceKey
302
- } from "./chunk-GRCDRKII.js";
285
+ } from "./chunk-WS3NZZQQ.js";
303
286
  import {
304
287
  MODEL_PRICING,
305
288
  MetricsCollector,
@@ -338,7 +321,7 @@ import {
338
321
  scoreTraceInsightReadiness,
339
322
  tokenizeDomainWords,
340
323
  traceAnalystOnRunComplete
341
- } from "./chunk-QFQZ3U3X.js";
324
+ } from "./chunk-OCFJACJU.js";
342
325
  import "./chunk-H5UD2323.js";
343
326
  import {
344
327
  extractUsage,
@@ -403,14 +386,26 @@ import {
403
386
  llmSpanFromProvider
404
387
  } from "./chunk-VQMK5FMP.js";
405
388
  import {
389
+ AGENT_PROFILE_KINDS,
390
+ AgentProfileCellValidationError,
406
391
  RunRecordValidationError,
392
+ agentProfileCellHashMaterial,
393
+ agentProfileCellKey,
394
+ assertRunAgentProfileCell,
395
+ buildAgentInterfaceProfileCell,
396
+ buildAgentProfileCell,
397
+ groupRunsByAgentProfileCell,
407
398
  isRunRecord,
408
399
  modelHasSnapshot,
409
400
  parseRunRecordSafe,
401
+ requireAgentProfileCell,
410
402
  resolveRunCostProvenance,
411
403
  roundTripRunRecord,
412
- validateRunRecord
413
- } from "./chunk-LOW3U7JZ.js";
404
+ toAgentProfileJson,
405
+ validateAgentProfileCell,
406
+ validateRunRecord,
407
+ verifyAgentProfileCell
408
+ } from "./chunk-IILEIWGW.js";
414
409
  import {
415
410
  FAILURE_CLASSES,
416
411
  TRACE_SCHEMA_VERSION,
@@ -420,20 +415,6 @@ import {
420
415
  isSandboxSpan,
421
416
  isToolSpan
422
417
  } from "./chunk-MA6HLL3S.js";
423
- import {
424
- AGENT_PROFILE_KINDS,
425
- AgentProfileCellValidationError,
426
- agentProfileCellHashMaterial,
427
- agentProfileCellKey,
428
- assertRunAgentProfileCell,
429
- buildAgentInterfaceProfileCell,
430
- buildAgentProfileCell,
431
- groupRunsByAgentProfileCell,
432
- requireAgentProfileCell,
433
- toAgentProfileJson,
434
- validateAgentProfileCell,
435
- verifyAgentProfileCell
436
- } from "./chunk-GC4ATIKK.js";
437
418
  import {
438
419
  canonicalize,
439
420
  evaluateHypothesis,
@@ -10064,414 +10045,6 @@ function isOtelConfigured() {
10064
10045
  return !!(typeof process !== "undefined" && process.env.OTEL_EXPORTER_OTLP_ENDPOINT);
10065
10046
  }
10066
10047
 
10067
- // src/traced-analyst.ts
10068
- async function tracedAnalyzeTraces(input, options, traceOpts) {
10069
- const parentSpan = await traceOpts.emitter.span({
10070
- kind: "custom",
10071
- name: "analyst:analyze-traces",
10072
- parentSpanId: traceOpts.parentSpanId,
10073
- attributes: {
10074
- "analyst.question_length": input.question.length,
10075
- "analyst.max_turns": options.maxTurns ?? 12,
10076
- "analyst.max_subqueries": options.maxSubqueries ?? 4,
10077
- "eval.phase": "analyst"
10078
- }
10079
- });
10080
- const originalOnTurn = options.onTurn;
10081
- const wrappedOptions = {
10082
- ...options,
10083
- onTurn: async (turn) => {
10084
- const turnSpan = await traceOpts.emitter.span({
10085
- kind: "custom",
10086
- name: `analyst:turn-${turn.turn}`,
10087
- parentSpanId: parentSpan.span.spanId,
10088
- attributes: {
10089
- "analyst.stage": turn.stage,
10090
- "analyst.turn": turn.turn,
10091
- "analyst.is_error": turn.isError,
10092
- "analyst.code_length": turn.code.length,
10093
- "analyst.output_length": turn.output.length,
10094
- "eval.phase": "analyst"
10095
- }
10096
- });
10097
- if (turn.isError) {
10098
- await turnSpan.fail("Turn produced an error");
10099
- } else {
10100
- await turnSpan.end();
10101
- }
10102
- if (originalOnTurn) await originalOnTurn(turn);
10103
- }
10104
- };
10105
- try {
10106
- const result = await analyzeTraces(input, wrappedOptions);
10107
- await parentSpan.end({
10108
- attributes: {
10109
- "analyst.question_length": input.question.length,
10110
- "analyst.turn_count": result.turnCount,
10111
- "analyst.finding_count": result.findings.length,
10112
- "analyst.answer_length": result.answer.length,
10113
- "eval.phase": "analyst"
10114
- }
10115
- });
10116
- return result;
10117
- } catch (err) {
10118
- await parentSpan.fail(err instanceof Error ? err : String(err));
10119
- throw err;
10120
- }
10121
- }
10122
-
10123
- // src/traced-judges.ts
10124
- function traceJudge(judge, judgeName, opts) {
10125
- return async (tc, input) => {
10126
- const span = await opts.emitter.span({
10127
- kind: "llm",
10128
- name: `judge:${judgeName}`,
10129
- parentSpanId: opts.parentSpanId,
10130
- attributes: {
10131
- "judge.name": judgeName,
10132
- "eval.phase": "judge"
10133
- }
10134
- });
10135
- try {
10136
- const scores2 = await judge(tc, input);
10137
- const composite = scores2.length > 0 ? scores2.reduce((sum4, s) => sum4 + s.score, 0) / scores2.length : 0;
10138
- await span.end({
10139
- attributes: {
10140
- "judge.name": judgeName,
10141
- "judge.composite_score": composite,
10142
- "judge.dimension_count": scores2.length,
10143
- "eval.phase": "judge"
10144
- }
10145
- });
10146
- return scores2;
10147
- } catch (err) {
10148
- await span.fail(err instanceof Error ? err : String(err));
10149
- throw err;
10150
- }
10151
- };
10152
- }
10153
- function traceJudgeEnsemble(judges, judgeNames, opts) {
10154
- return async (tc, input) => {
10155
- const ensembleSpan = await opts.emitter.span({
10156
- kind: "custom",
10157
- name: "judge:ensemble",
10158
- parentSpanId: opts.parentSpanId,
10159
- attributes: {
10160
- "judge.ensemble_size": judges.length,
10161
- "eval.phase": "judge"
10162
- }
10163
- });
10164
- try {
10165
- const allScores = [];
10166
- let failedJudges = 0;
10167
- for (let i = 0; i < judges.length; i++) {
10168
- const judge = judges[i];
10169
- const name = judgeNames[i] ?? `judge_${i}`;
10170
- const tracedFn = traceJudge(judge, name, {
10171
- emitter: opts.emitter,
10172
- parentSpanId: ensembleSpan.span.spanId
10173
- });
10174
- try {
10175
- const scores2 = await tracedFn(tc, input);
10176
- allScores.push(...scores2);
10177
- } catch (err) {
10178
- if (!(err instanceof JudgeParseError)) throw err;
10179
- failedJudges++;
10180
- }
10181
- }
10182
- const composite = allScores.length > 0 ? allScores.reduce((sum4, s) => sum4 + s.score, 0) / allScores.length : 0;
10183
- await ensembleSpan.end({
10184
- attributes: {
10185
- "judge.ensemble_size": judges.length,
10186
- "judge.composite_score": composite,
10187
- "judge.total_dimensions": allScores.length,
10188
- "judge.failed_judges": failedJudges,
10189
- "eval.phase": "judge"
10190
- }
10191
- });
10192
- return allScores;
10193
- } catch (err) {
10194
- await ensembleSpan.fail(err instanceof Error ? err : String(err));
10195
- throw err;
10196
- }
10197
- };
10198
- }
10199
-
10200
- // src/campaign/distillation/agreement-judge.ts
10201
- var AGREEMENT_DIM = "agreement";
10202
- function buildAgreementJudge(options) {
10203
- const name = options.name ?? "gold-agreement";
10204
- const goldOnly = options.goldOnly ?? true;
10205
- const declaredDims = options.dimensionKeys ?? [AGREEMENT_DIM];
10206
- return {
10207
- name,
10208
- dimensions: declaredDims.map((key) => ({
10209
- key,
10210
- description: `Per-field agreement between the produced label and the gold label on '${key}'`
10211
- })),
10212
- appliesTo: goldOnly ? (scenario) => scenario.kind === "gold" : void 0,
10213
- score({ artifact, scenario }) {
10214
- const { score, dimensions } = options.compareLabels(artifact, scenario.label);
10215
- if (!Number.isFinite(score) || score < 0 || score > 1) {
10216
- throw new Error(
10217
- `buildAgreementJudge: comparator returned out-of-range score ${score} for scenario '${scenario.id}' (must be in [0,1])`
10218
- );
10219
- }
10220
- const outDims = { [AGREEMENT_DIM]: score, ...dimensions };
10221
- const weakest = Object.entries(dimensions).sort((a, b) => a[1] - b[1])[0];
10222
- const notes = weakest ? `agreement ${score.toFixed(3)}; weakest field '${weakest[0]}' (${weakest[1].toFixed(3)})` : `agreement ${score.toFixed(3)}`;
10223
- return { composite: score, dimensions: outDims, notes };
10224
- }
10225
- };
10226
- }
10227
- function fieldAgreement(spec) {
10228
- const categorical = spec.categorical ?? [];
10229
- const array = spec.array ?? [];
10230
- if (categorical.length === 0 && array.length === 0) {
10231
- throw new Error("fieldAgreement: at least one categorical or array field is required");
10232
- }
10233
- return (produced, gold) => {
10234
- const p = produced ?? {};
10235
- const g = gold ?? {};
10236
- const dimensions = {};
10237
- for (const field of categorical) {
10238
- dimensions[field] = categoricalAgreement(p[field], g[field]);
10239
- }
10240
- for (const field of array) {
10241
- dimensions[field] = jaccard(asArray(p[field]), asArray(g[field]));
10242
- }
10243
- const values = Object.values(dimensions);
10244
- const score = values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length;
10245
- return { score, dimensions };
10246
- };
10247
- }
10248
- function categoricalAgreement(produced, gold) {
10249
- if (produced === void 0 && gold === void 0) return 1;
10250
- return normalizeScalar(produced) === normalizeScalar(gold) ? 1 : 0;
10251
- }
10252
- function normalizeScalar(value) {
10253
- if (value === void 0) return "__undefined__";
10254
- if (value === null) return "__null__";
10255
- return JSON.stringify(value);
10256
- }
10257
- function asArray(value) {
10258
- if (Array.isArray(value)) return value;
10259
- if (value === void 0 || value === null) return [];
10260
- return [value];
10261
- }
10262
- function jaccard(a, b) {
10263
- const sa = new Set(a.map((x) => JSON.stringify(x)));
10264
- const sb = new Set(b.map((x) => JSON.stringify(x)));
10265
- if (sa.size === 0 && sb.size === 0) return 1;
10266
- let inter = 0;
10267
- for (const x of sa) if (sb.has(x)) inter++;
10268
- const union = sa.size + sb.size - inter;
10269
- return union === 0 ? 1 : inter / union;
10270
- }
10271
-
10272
- // src/campaign/distillation/gold-scenarios.ts
10273
- import { readFileSync as readFileSync5 } from "fs";
10274
- function loadGoldScenarios(jsonlPath) {
10275
- const text = readFileSync5(jsonlPath, "utf8");
10276
- return parseGoldJsonl(text, jsonlPath);
10277
- }
10278
- function parseGoldJsonl(text, sourceLabel = "<inline>") {
10279
- const out = [];
10280
- const lines = text.split("\n");
10281
- for (let i = 0; i < lines.length; i++) {
10282
- const raw = lines[i].trim();
10283
- if (raw.length === 0) continue;
10284
- let parsed;
10285
- try {
10286
- parsed = JSON.parse(raw);
10287
- } catch (err) {
10288
- throw new Error(
10289
- `loadGoldScenarios: ${sourceLabel}:${i + 1} is not valid JSON \u2014 ${err instanceof Error ? err.message : String(err)}`
10290
- );
10291
- }
10292
- const rawId = parsed.scenarioId ?? parsed.id;
10293
- if (typeof rawId !== "string" || rawId.length === 0) {
10294
- throw new Error(
10295
- `loadGoldScenarios: ${sourceLabel}:${i + 1} missing string \`scenarioId\`/\`id\``
10296
- );
10297
- }
10298
- const id = rawId.replace(/:/g, "__");
10299
- if (parsed.input === void 0) {
10300
- throw new Error(`loadGoldScenarios: ${sourceLabel}:${i + 1} (${rawId}) missing \`input\``);
10301
- }
10302
- if (parsed.label === void 0) {
10303
- throw new Error(`loadGoldScenarios: ${sourceLabel}:${i + 1} (${rawId}) missing \`label\``);
10304
- }
10305
- const scenario = {
10306
- id,
10307
- kind: "gold",
10308
- input: parsed.input,
10309
- label: parsed.label
10310
- };
10311
- const tags = [];
10312
- if (id !== rawId) tags.push(`gold-id:${rawId}`);
10313
- if (parsed.split !== void 0) tags.push(`split:${parsed.split}`);
10314
- if (tags.length > 0) scenario.tags = tags;
10315
- out.push(scenario);
10316
- }
10317
- if (out.length === 0) {
10318
- throw new Error(`loadGoldScenarios: ${sourceLabel} contained no gold records`);
10319
- }
10320
- return out;
10321
- }
10322
- function splitGold(scenarios, options = {}) {
10323
- const testEveryNth = options.testEveryNth ?? 4;
10324
- if (!Number.isInteger(testEveryNth) || testEveryNth < 2) {
10325
- throw new Error("splitGold: testEveryNth must be an integer \u2265 2 (else train or test is empty)");
10326
- }
10327
- const train = [];
10328
- const test = [];
10329
- let implicitIndex = 0;
10330
- for (const scenario of scenarios) {
10331
- const explicit = explicitSplit(scenario);
10332
- if (explicit === "train") {
10333
- train.push(scenario);
10334
- } else if (explicit === "test") {
10335
- test.push(scenario);
10336
- } else {
10337
- if (implicitIndex % testEveryNth === 0) test.push(scenario);
10338
- else train.push(scenario);
10339
- implicitIndex += 1;
10340
- }
10341
- }
10342
- return { train, test };
10343
- }
10344
- function explicitSplit(scenario) {
10345
- for (const tag of scenario.tags ?? []) {
10346
- if (tag === "split:train") return "train";
10347
- if (tag === "split:test") return "test";
10348
- }
10349
- return void 0;
10350
- }
10351
-
10352
- // src/campaign/distillation/run-distillation.ts
10353
- async function runDistillation(opts) {
10354
- if (opts.train.length === 0) throw new Error("runDistillation: train split is empty");
10355
- if (opts.holdout.length === 0) throw new Error("runDistillation: holdout split is empty");
10356
- const chat = createChatClient(opts.llm);
10357
- const render = opts.renderStudentPrompt ?? defaultRenderStudentPrompt;
10358
- const parse = opts.parseStudentLabel ?? defaultParseStudentLabel;
10359
- const runDir = opts.runDir ?? `.evolve/distillation/${Date.now()}`;
10360
- const studentTemperature = opts.studentTemperature ?? 0;
10361
- const studentMaxTokens = opts.studentMaxTokens ?? 1024;
10362
- const proposer = gepaProposer({
10363
- llm: opts.reflectionLlm,
10364
- model: opts.optimizerModel,
10365
- target: "a cheap single-shot analyst system prompt that reproduces an expensive workflow gold verdict",
10366
- mutationPrimitives: opts.mutationPrimitives ?? DEFAULT_MUTATION_PRIMITIVES2,
10367
- constraints: opts.constraints
10368
- });
10369
- const gate = opts.gate ?? heldOutGate({
10370
- scenarios: opts.holdout,
10371
- deltaThreshold: opts.deltaThreshold ?? 0
10372
- });
10373
- const loop = await runImprovementLoop({
10374
- baselineSurface: opts.baselinePrompt,
10375
- scenarios: opts.train,
10376
- holdoutScenarios: opts.holdout,
10377
- judges: [opts.judge],
10378
- proposer,
10379
- gate,
10380
- autoOnPromote: "none",
10381
- // the loop NEVER opens a PR — the caller decides
10382
- populationSize: opts.populationSize ?? 4,
10383
- maxGenerations: opts.maxGenerations ?? 3,
10384
- reps: opts.reps ?? 1,
10385
- runDir,
10386
- // The student spends tokens; tracing must stay on (the proposer is wired and
10387
- // runImprovementLoop refuses tracing='off' with a proposer).
10388
- tracing: "on",
10389
- dispatchWithSurface: async (surface, scenario, ctx) => {
10390
- const prompt = render({
10391
- surface: typeof surface === "string" ? surface : JSON.stringify(surface),
10392
- input: scenario.input,
10393
- scenarioId: scenario.id
10394
- });
10395
- const request = {
10396
- model: opts.studentModel,
10397
- messages: prompt,
10398
- jsonMode: true,
10399
- temperature: studentTemperature,
10400
- maxTokens: studentMaxTokens
10401
- };
10402
- const paid = await ctx.cost.runPaidCall({
10403
- actor: "distillation-student",
10404
- model: opts.studentModel,
10405
- maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maxRetries: chat.maximumAttempts }),
10406
- execute: (signal, callId) => chat.chat(request, { signal, idempotencyKey: callId }),
10407
- receipt: costReceiptFromLlm,
10408
- receiptFromError: costReceiptFromLlmError
10409
- });
10410
- if (!paid.succeeded) throw paid.error;
10411
- return parse(paid.value.content, scenario.id);
10412
- }
10413
- });
10414
- const winnerPrompt = typeof loop.winnerSurface === "string" ? loop.winnerSurface : opts.baselinePrompt;
10415
- const baseline = campaignMeanComposite(loop.baselineOnHoldout);
10416
- const winner = campaignMeanComposite(loop.winnerOnHoldout);
10417
- return {
10418
- ...loop,
10419
- winnerPrompt,
10420
- holdoutAgreement: { baseline, winner, delta: winner - baseline }
10421
- };
10422
- }
10423
- var DEFAULT_MUTATION_PRIMITIVES2 = [
10424
- "Add an explicit output-schema instruction so the model emits exactly the gold label fields as JSON.",
10425
- "Add a one-line decision rule for each verdict field the student keeps getting wrong.",
10426
- "Add a worked example mapping a representative input to its correct gold label.",
10427
- "Tighten ambiguous phrasing that lets the student hedge instead of committing to a verdict.",
10428
- "Add a guardrail that forces the student to set boolean risk flags (e.g. leak risk) when the triggering condition is present."
10429
- ];
10430
- function defaultRenderStudentPrompt(args) {
10431
- return [
10432
- { role: "system", content: args.surface },
10433
- {
10434
- role: "user",
10435
- content: `Input:
10436
- ${stableStringify(args.input)}
10437
-
10438
- Respond with ONLY a single JSON object \u2014 the verdict. No prose, no code fences.`
10439
- }
10440
- ];
10441
- }
10442
- function defaultParseStudentLabel(rawContent, scenarioId) {
10443
- const stripped = stripFence(rawContent).trim();
10444
- if (stripped.length === 0) {
10445
- throw new Error(`distillation student returned empty output for scenario '${scenarioId}'`);
10446
- }
10447
- try {
10448
- return JSON.parse(stripped);
10449
- } catch (err) {
10450
- throw new Error(
10451
- `distillation student returned non-JSON for scenario '${scenarioId}': ${err instanceof Error ? err.message : String(err)} \u2014 raw: ${stripped.slice(0, 200)}`
10452
- );
10453
- }
10454
- }
10455
- function stripFence(text) {
10456
- const fenced = /```(?:json)?\s*([\s\S]*?)\s*```/.exec(text);
10457
- return fenced ? fenced[1] ?? text : text;
10458
- }
10459
- function stableStringify(value) {
10460
- return JSON.stringify(value, replacerSortKeys(), 2);
10461
- }
10462
- function replacerSortKeys() {
10463
- return (_key, value) => {
10464
- if (value && typeof value === "object" && !Array.isArray(value)) {
10465
- const sorted = {};
10466
- for (const k of Object.keys(value).sort()) {
10467
- sorted[k] = value[k];
10468
- }
10469
- return sorted;
10470
- }
10471
- return value;
10472
- };
10473
- }
10474
-
10475
10048
  // src/profile/index.ts
10476
10049
  var profile_exports = {};
10477
10050
  __export(profile_exports, {
@@ -10613,6 +10186,139 @@ function sectionHash(section) {
10613
10186
  return surfaceContentHash(JSON.stringify({ title: section.title, body: section.body }));
10614
10187
  }
10615
10188
 
10189
+ // src/traced-analyst.ts
10190
+ async function tracedAnalyzeTraces(input, options, traceOpts) {
10191
+ const parentSpan = await traceOpts.emitter.span({
10192
+ kind: "custom",
10193
+ name: "analyst:analyze-traces",
10194
+ parentSpanId: traceOpts.parentSpanId,
10195
+ attributes: {
10196
+ "analyst.question_length": input.question.length,
10197
+ "analyst.max_turns": options.maxTurns ?? 12,
10198
+ "analyst.max_subqueries": options.maxSubqueries ?? 4,
10199
+ "eval.phase": "analyst"
10200
+ }
10201
+ });
10202
+ const originalOnTurn = options.onTurn;
10203
+ const wrappedOptions = {
10204
+ ...options,
10205
+ onTurn: async (turn) => {
10206
+ const turnSpan = await traceOpts.emitter.span({
10207
+ kind: "custom",
10208
+ name: `analyst:turn-${turn.turn}`,
10209
+ parentSpanId: parentSpan.span.spanId,
10210
+ attributes: {
10211
+ "analyst.stage": turn.stage,
10212
+ "analyst.turn": turn.turn,
10213
+ "analyst.is_error": turn.isError,
10214
+ "analyst.code_length": turn.code.length,
10215
+ "analyst.output_length": turn.output.length,
10216
+ "eval.phase": "analyst"
10217
+ }
10218
+ });
10219
+ if (turn.isError) {
10220
+ await turnSpan.fail("Turn produced an error");
10221
+ } else {
10222
+ await turnSpan.end();
10223
+ }
10224
+ if (originalOnTurn) await originalOnTurn(turn);
10225
+ }
10226
+ };
10227
+ try {
10228
+ const result = await analyzeTraces(input, wrappedOptions);
10229
+ await parentSpan.end({
10230
+ attributes: {
10231
+ "analyst.question_length": input.question.length,
10232
+ "analyst.turn_count": result.turnCount,
10233
+ "analyst.finding_count": result.findings.length,
10234
+ "analyst.answer_length": result.answer.length,
10235
+ "eval.phase": "analyst"
10236
+ }
10237
+ });
10238
+ return result;
10239
+ } catch (err) {
10240
+ await parentSpan.fail(err instanceof Error ? err : String(err));
10241
+ throw err;
10242
+ }
10243
+ }
10244
+
10245
+ // src/traced-judges.ts
10246
+ function traceJudge(judge, judgeName, opts) {
10247
+ return async (tc, input) => {
10248
+ const span = await opts.emitter.span({
10249
+ kind: "llm",
10250
+ name: `judge:${judgeName}`,
10251
+ parentSpanId: opts.parentSpanId,
10252
+ attributes: {
10253
+ "judge.name": judgeName,
10254
+ "eval.phase": "judge"
10255
+ }
10256
+ });
10257
+ try {
10258
+ const scores2 = await judge(tc, input);
10259
+ const composite = scores2.length > 0 ? scores2.reduce((sum4, s) => sum4 + s.score, 0) / scores2.length : 0;
10260
+ await span.end({
10261
+ attributes: {
10262
+ "judge.name": judgeName,
10263
+ "judge.composite_score": composite,
10264
+ "judge.dimension_count": scores2.length,
10265
+ "eval.phase": "judge"
10266
+ }
10267
+ });
10268
+ return scores2;
10269
+ } catch (err) {
10270
+ await span.fail(err instanceof Error ? err : String(err));
10271
+ throw err;
10272
+ }
10273
+ };
10274
+ }
10275
+ function traceJudgeEnsemble(judges, judgeNames, opts) {
10276
+ return async (tc, input) => {
10277
+ const ensembleSpan = await opts.emitter.span({
10278
+ kind: "custom",
10279
+ name: "judge:ensemble",
10280
+ parentSpanId: opts.parentSpanId,
10281
+ attributes: {
10282
+ "judge.ensemble_size": judges.length,
10283
+ "eval.phase": "judge"
10284
+ }
10285
+ });
10286
+ try {
10287
+ const allScores = [];
10288
+ let failedJudges = 0;
10289
+ for (let i = 0; i < judges.length; i++) {
10290
+ const judge = judges[i];
10291
+ const name = judgeNames[i] ?? `judge_${i}`;
10292
+ const tracedFn = traceJudge(judge, name, {
10293
+ emitter: opts.emitter,
10294
+ parentSpanId: ensembleSpan.span.spanId
10295
+ });
10296
+ try {
10297
+ const scores2 = await tracedFn(tc, input);
10298
+ allScores.push(...scores2);
10299
+ } catch (err) {
10300
+ if (!(err instanceof JudgeParseError)) throw err;
10301
+ failedJudges++;
10302
+ }
10303
+ }
10304
+ const composite = allScores.length > 0 ? allScores.reduce((sum4, s) => sum4 + s.score, 0) / allScores.length : 0;
10305
+ await ensembleSpan.end({
10306
+ attributes: {
10307
+ "judge.ensemble_size": judges.length,
10308
+ "judge.composite_score": composite,
10309
+ "judge.total_dimensions": allScores.length,
10310
+ "judge.failed_judges": failedJudges,
10311
+ "eval.phase": "judge"
10312
+ }
10313
+ });
10314
+ return allScores;
10315
+ } catch (err) {
10316
+ await ensembleSpan.fail(err instanceof Error ? err : String(err));
10317
+ throw err;
10318
+ }
10319
+ };
10320
+ }
10321
+
10616
10322
  // src/cost-report.ts
10617
10323
  function costReport(ledger) {
10618
10324
  const summary = ledger.summary();
@@ -10733,13 +10439,13 @@ function verifyAttestation(report, attested) {
10733
10439
  }
10734
10440
 
10735
10441
  // src/product-benchmark/index.ts
10736
- import { existsSync as existsSync6, readFileSync as readFileSync7, statSync as statSync3 } from "fs";
10442
+ import { existsSync as existsSync6, readFileSync as readFileSync6, statSync as statSync3 } from "fs";
10737
10443
  import { dirname as dirname4, join as join5 } from "path";
10738
10444
 
10739
10445
  // src/product-benchmark/export.ts
10740
10446
  import { spawnSync as spawnSync2 } from "child_process";
10741
10447
  import { createHash } from "crypto";
10742
- import { cpSync, existsSync as existsSync5, mkdirSync as mkdirSync3, readFileSync as readFileSync6, writeFileSync } from "fs";
10448
+ import { cpSync, existsSync as existsSync5, mkdirSync as mkdirSync3, readFileSync as readFileSync5, writeFileSync } from "fs";
10743
10449
  import { basename as basename2, dirname as dirname3, isAbsolute, join as join4, relative, resolve } from "path";
10744
10450
  var productBenchmarkMutableSurfaces = [
10745
10451
  "prompt",
@@ -10787,19 +10493,19 @@ function productBenchmarkRepoIdentity() {
10787
10493
  };
10788
10494
  }
10789
10495
  function packageVersion(name) {
10790
- const pkg = JSON.parse(readFileSync6(resolve("package.json"), "utf8"));
10496
+ const pkg = JSON.parse(readFileSync5(resolve("package.json"), "utf8"));
10791
10497
  if (pkg.name === name && pkg.version) return pkg.version;
10792
10498
  const declared = pkg.dependencies?.[name] ?? pkg.devDependencies?.[name];
10793
10499
  if (declared) return declared;
10794
10500
  const installed = resolve("node_modules", name, "package.json");
10795
10501
  if (existsSync5(installed)) {
10796
- const installedPkg = JSON.parse(readFileSync6(installed, "utf8"));
10502
+ const installedPkg = JSON.parse(readFileSync5(installed, "utf8"));
10797
10503
  if (installedPkg.version) return installedPkg.version;
10798
10504
  }
10799
10505
  return "unknown";
10800
10506
  }
10801
10507
  function readRunRecords(path) {
10802
- const lines = readFileSync6(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
10508
+ const lines = readFileSync5(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
10803
10509
  return lines.map((line, index) => {
10804
10510
  let parsed;
10805
10511
  try {
@@ -11450,7 +11156,7 @@ function productBenchmarkIntegrityFailures(record) {
11450
11156
  return failures;
11451
11157
  }
11452
11158
  function readProductBenchmarkRecords(path) {
11453
- const lines = readFileSync7(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
11159
+ const lines = readFileSync6(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
11454
11160
  const records = [];
11455
11161
  for (const [index, line] of lines.entries()) {
11456
11162
  try {
@@ -11463,7 +11169,7 @@ function readProductBenchmarkRecords(path) {
11463
11169
  }
11464
11170
  function readProductBenchmarkManifest(path) {
11465
11171
  try {
11466
- return validateProductBenchmarkManifest(JSON.parse(readFileSync7(path, "utf8")));
11172
+ return validateProductBenchmarkManifest(JSON.parse(readFileSync6(path, "utf8")));
11467
11173
  } catch (err) {
11468
11174
  wrapValidationError(path, err);
11469
11175
  }
@@ -11666,11 +11372,7 @@ export {
11666
11372
  OTEL_AGENT_EVAL_SCOPE,
11667
11373
  OUTPUT_VALUE,
11668
11374
  OtlpFileTraceStore,
11669
- POLICY_EDIT_AXES,
11670
- POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
11671
- POLICY_EDIT_TARGET_SURFACES,
11672
11375
  PairwiseSteeringOptimizer,
11673
- PolicyEditValidationError,
11674
11376
  ProductClient,
11675
11377
  PromptRegistry,
11676
11378
  REDACTION_VERSION,
@@ -11716,7 +11418,6 @@ export {
11716
11418
  ValidationError,
11717
11419
  VerificationError,
11718
11420
  acquisitionPlansForKnowledgeGaps,
11719
- admitPolicyEdit,
11720
11421
  adversarialJudge,
11721
11422
  agentProfileCellHashMaterial,
11722
11423
  agentProfileCellKey,
@@ -11736,7 +11437,6 @@ export {
11736
11437
  analyzeTraces,
11737
11438
  appendScorecard,
11738
11439
  applyLlmSpanOtlpAttributes,
11739
- applyPolicyEditToSurface,
11740
11440
  applyToolSpanOtlpAttributes,
11741
11441
  argHash,
11742
11442
  asNumber,
@@ -11770,7 +11470,6 @@ export {
11770
11470
  bootstrapCi,
11771
11471
  buildAgentInterfaceProfileCell,
11772
11472
  buildAgentProfileCell,
11773
- buildAgreementJudge,
11774
11473
  buildDefaultAnalystRegistry,
11775
11474
  buildDriverSystemPrompt,
11776
11475
  buildProductBenchmarkManifest,
@@ -11800,6 +11499,7 @@ export {
11800
11499
  clamp01,
11801
11500
  classifyFailure,
11802
11501
  classifyTreatment,
11502
+ claudeCodeSupervisorRunReader,
11803
11503
  cliffsDelta,
11804
11504
  clusteredPairedBinary,
11805
11505
  codeExecutionJudge,
@@ -11817,7 +11517,6 @@ export {
11817
11517
  composeValidators,
11818
11518
  computeExperimentStats,
11819
11519
  computeFindingId,
11820
- computePolicyEditId,
11821
11520
  computeToolUseMetrics,
11822
11521
  computeTraceMetrics,
11823
11522
  confidenceInterval,
@@ -11864,10 +11563,8 @@ export {
11864
11563
  defaultBlendWeights,
11865
11564
  defaultIsMaterial,
11866
11565
  defaultJudges,
11867
- defaultParseStudentLabel,
11868
11566
  defaultProviderRedactor,
11869
11567
  defaultReferenceReplayMatcher,
11870
- defaultRenderStudentPrompt,
11871
11568
  defaultTraceInsightPanel,
11872
11569
  deployGateLayer,
11873
11570
  describeTraceInsightScope,
@@ -11907,7 +11604,6 @@ export {
11907
11604
  feedbackTrajectoriesToOptimizerRows,
11908
11605
  feedbackTrajectoryToDatasetScenario,
11909
11606
  feedbackTrajectoryToOptimizerRow,
11910
- fieldAgreement,
11911
11607
  fileContains,
11912
11608
  fileExists,
11913
11609
  fileExperimentStore,
@@ -11965,7 +11661,6 @@ export {
11965
11661
  isLlmSpan,
11966
11662
  isModelPriced,
11967
11663
  isOtelConfigured,
11968
- isPolicyEdit,
11969
11664
  isRetrievalSpan,
11970
11665
  isRolloutLine,
11971
11666
  isRunRecord,
@@ -11991,15 +11686,12 @@ export {
11991
11686
  llmJudge,
11992
11687
  llmSpanFromProvider,
11993
11688
  llmSpans,
11994
- loadGoldScenarios,
11995
11689
  loadScorecard,
11996
11690
  loadScorerFromGrader,
11997
11691
  localCommandRunner,
11998
11692
  lowercaseMutator,
11999
11693
  makeEvalTools,
12000
11694
  makeFinding,
12001
- makePolicyEdit,
12002
- makePolicyEditCandidateRecord,
12003
11695
  mannWhitneyU,
12004
11696
  mapConcurrent,
12005
11697
  matchGoldens,
@@ -12040,7 +11732,6 @@ export {
12040
11732
  paretoFrontierWithCrowding,
12041
11733
  parseCorrectnessResponse,
12042
11734
  parseFeedbackTrajectoriesJsonl,
12043
- parseGoldJsonl,
12044
11735
  parseReflectionResponse,
12045
11736
  parseRunRecordSafe,
12046
11737
  parseRuntimeTrajectoryHookEvent,
@@ -12051,8 +11742,6 @@ export {
12051
11742
  pearsonR,
12052
11743
  pixelDeltaRatio,
12053
11744
  planTraceInsightQuestions,
12054
- policyEditFromFinding,
12055
- policyEditsFromFindings,
12056
11745
  politenessPrefixMutator,
12057
11746
  positionalBias,
12058
11747
  preflightModels,
@@ -12070,6 +11759,7 @@ export {
12070
11759
  providerFromBaseUrl,
12071
11760
  pytestTestParser,
12072
11761
  ranks,
11762
+ readClaudeCodeSupervisorRun,
12073
11763
  readOtlpStatus,
12074
11764
  readProductBenchmarkManifest,
12075
11765
  readProductBenchmarkRecords,
@@ -12114,7 +11804,6 @@ export {
12114
11804
  runBehavioralCanaries,
12115
11805
  runCanaries,
12116
11806
  runCounterfactual,
12117
- runDistillation,
12118
11807
  runE2EWorkflow,
12119
11808
  runEvalCampaign,
12120
11809
  runExpectations,
@@ -12140,7 +11829,6 @@ export {
12140
11829
  scoreContinuity,
12141
11830
  scoreFromEvals,
12142
11831
  scoreKnowledgeReadiness,
12143
- scorePolicyEditReadiness,
12144
11832
  scorePrReviewComments,
12145
11833
  scorePrReviewSource,
12146
11834
  scoreRedTeamOutput,
@@ -12155,7 +11843,6 @@ export {
12155
11843
  showMeasured,
12156
11844
  signManifest,
12157
11845
  spearmanR,
12158
- splitGold,
12159
11846
  statusAdvanced,
12160
11847
  stopOnNoProgress,
12161
11848
  stopOnRepeatedAction,
@@ -12193,8 +11880,6 @@ export {
12193
11880
  urlContains,
12194
11881
  userQuestionsForKnowledgeGaps,
12195
11882
  validateAgentProfileCell,
12196
- validatePolicyEdit,
12197
- validatePolicyEditCandidateRecord,
12198
11883
  validateProductBenchmarkManifest,
12199
11884
  validateProductBenchmarkRecord,
12200
11885
  validateProductBenchmarkRun,