@tangle-network/agent-eval 0.108.1 → 0.110.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (175) hide show
  1. package/dist/analyst/index.d.ts +10 -12
  2. package/dist/analyst/index.js +8 -11
  3. package/dist/analyst/index.js.map +1 -1
  4. package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
  5. package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
  6. package/dist/belief-state/index.d.ts +6 -6
  7. package/dist/benchmarks/index.d.ts +4 -4
  8. package/dist/benchmarks/index.js +7 -8
  9. package/dist/builder-eval/index.d.ts +4 -4
  10. package/dist/builder-eval/index.js +1 -2
  11. package/dist/builder-eval/index.js.map +1 -1
  12. package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
  13. package/dist/campaign/index.d.ts +161 -20
  14. package/dist/campaign/index.js +15 -8
  15. package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
  16. package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
  17. package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
  18. package/dist/chunk-7NX6ZSBG.js.map +1 -0
  19. package/dist/{chunk-OVPVM4JC.js → chunk-GTERJI6Q.js} +4 -4
  20. package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
  21. package/dist/chunk-IMWDSFUM.js.map +1 -0
  22. package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
  23. package/dist/chunk-MHNQWM4I.js.map +1 -0
  24. package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
  25. package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
  26. package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
  27. package/dist/chunk-PLOMR3HP.js.map +1 -0
  28. package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
  29. package/dist/{chunk-6SKVFBTR.js → chunk-RNB2NICW.js} +115 -13
  30. package/dist/chunk-RNB2NICW.js.map +1 -0
  31. package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
  32. package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
  33. package/dist/chunk-XRGOKCMO.js.map +1 -0
  34. package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
  35. package/dist/contract/index.d.ts +20 -23
  36. package/dist/contract/index.js +11 -13
  37. package/dist/contract/index.js.map +1 -1
  38. package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
  39. package/dist/control.d.ts +8 -9
  40. package/dist/control.js +6 -8
  41. package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
  42. package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
  43. package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
  44. package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
  45. package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
  46. package/dist/{gepa-B3x5Ulcv.d.ts → gepa-BUNP3606.d.ts} +143 -2
  47. package/dist/hosted/index.d.ts +7 -7
  48. package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
  49. package/dist/index.d.ts +645 -61
  50. package/dist/index.js +1282 -190
  51. package/dist/index.js.map +1 -1
  52. package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
  53. package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
  54. package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
  55. package/dist/meta-eval/index.d.ts +5 -5
  56. package/dist/meta-eval/index.js +1 -2
  57. package/dist/meta-eval/index.js.map +1 -1
  58. package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
  59. package/dist/multishot/index.d.ts +3 -3
  60. package/dist/openapi.json +1 -1
  61. package/dist/pipelines/index.d.ts +6 -7
  62. package/dist/pipelines/index.js +3 -6
  63. package/dist/pipelines/index.js.map +1 -1
  64. package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
  65. package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
  66. package/dist/{provenance-DdDhf6cg.d.ts → provenance-DMvsfknv.d.ts} +3 -5
  67. package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
  68. package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
  69. package/dist/reporting.d.ts +8 -8
  70. package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
  71. package/dist/rl.d.ts +568 -15
  72. package/dist/rl.js +4 -4
  73. package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
  74. package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
  75. package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
  76. package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
  77. package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
  78. package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
  79. package/dist/storyboard/index.d.ts +1 -1
  80. package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
  81. package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
  82. package/dist/traces.d.ts +54 -11
  83. package/dist/traces.js +25 -27
  84. package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
  85. package/dist/wire/index.d.ts +5 -6
  86. package/package.json +1 -71
  87. package/dist/adapters/http.d.ts +0 -142
  88. package/dist/adapters/http.js +0 -203
  89. package/dist/adapters/http.js.map +0 -1
  90. package/dist/adapters/langchain.d.ts +0 -95
  91. package/dist/adapters/langchain.js +0 -34
  92. package/dist/adapters/langchain.js.map +0 -1
  93. package/dist/adapters/otel.d.ts +0 -112
  94. package/dist/adapters/otel.js +0 -110
  95. package/dist/adapters/otel.js.map +0 -1
  96. package/dist/chunk-2OGPXHOB.js.map +0 -1
  97. package/dist/chunk-45EEMHTC.js +0 -35
  98. package/dist/chunk-45EEMHTC.js.map +0 -1
  99. package/dist/chunk-5BKGXME7.js +0 -65
  100. package/dist/chunk-5BKGXME7.js.map +0 -1
  101. package/dist/chunk-5PK3626Q.js.map +0 -1
  102. package/dist/chunk-6SK5VFYK.js +0 -100
  103. package/dist/chunk-6SK5VFYK.js.map +0 -1
  104. package/dist/chunk-6SKVFBTR.js.map +0 -1
  105. package/dist/chunk-DBDRR6GF.js.map +0 -1
  106. package/dist/chunk-DJWX3GVS.js +0 -81
  107. package/dist/chunk-DJWX3GVS.js.map +0 -1
  108. package/dist/chunk-FOUG2VVS.js +0 -855
  109. package/dist/chunk-FOUG2VVS.js.map +0 -1
  110. package/dist/chunk-JZXGWLK5.js.map +0 -1
  111. package/dist/chunk-K7QEIHHJ.js +0 -613
  112. package/dist/chunk-K7QEIHHJ.js.map +0 -1
  113. package/dist/chunk-KKHDIONI.js +0 -414
  114. package/dist/chunk-KKHDIONI.js.map +0 -1
  115. package/dist/chunk-KMPRBJK4.js +0 -74
  116. package/dist/chunk-KMPRBJK4.js.map +0 -1
  117. package/dist/chunk-Q2JRAWRI.js +0 -196
  118. package/dist/chunk-Q2JRAWRI.js.map +0 -1
  119. package/dist/chunk-RZTMDUO7.js +0 -49
  120. package/dist/chunk-RZTMDUO7.js.map +0 -1
  121. package/dist/chunk-STGVSCDH.js +0 -202
  122. package/dist/chunk-STGVSCDH.js.map +0 -1
  123. package/dist/chunk-YEHAEDUD.js.map +0 -1
  124. package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
  125. package/dist/corpus-eBVwhCp1.d.ts +0 -560
  126. package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
  127. package/dist/diagnose.d.ts +0 -252
  128. package/dist/diagnose.js +0 -382
  129. package/dist/diagnose.js.map +0 -1
  130. package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
  131. package/dist/governance/index.d.ts +0 -135
  132. package/dist/governance/index.js +0 -18
  133. package/dist/governance/index.js.map +0 -1
  134. package/dist/groundedness/index.d.ts +0 -112
  135. package/dist/groundedness/index.js +0 -77
  136. package/dist/groundedness/index.js.map +0 -1
  137. package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
  138. package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
  139. package/dist/knowledge/index.d.ts +0 -103
  140. package/dist/knowledge/index.js +0 -18
  141. package/dist/knowledge/index.js.map +0 -1
  142. package/dist/pareto-E-pembql.d.ts +0 -81
  143. package/dist/perf/index.d.ts +0 -123
  144. package/dist/perf/index.js +0 -18
  145. package/dist/perf/index.js.map +0 -1
  146. package/dist/prm/index.d.ts +0 -104
  147. package/dist/prm/index.js +0 -265
  148. package/dist/prm/index.js.map +0 -1
  149. package/dist/product-benchmark/index.d.ts +0 -247
  150. package/dist/product-benchmark/index.js +0 -37
  151. package/dist/product-benchmark/index.js.map +0 -1
  152. package/dist/red-team-KmmiqBlY.d.ts +0 -63
  153. package/dist/redact-B40YG2M_.d.ts +0 -45
  154. package/dist/rubric-Cc6UHvUb.d.ts +0 -73
  155. package/dist/run-critic-CmMf05uV.d.ts +0 -56
  156. package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
  157. package/dist/telemetry/file.d.ts +0 -19
  158. package/dist/telemetry/file.js +0 -45
  159. package/dist/telemetry/file.js.map +0 -1
  160. package/dist/telemetry/index.d.ts +0 -38
  161. package/dist/telemetry/index.js +0 -130
  162. package/dist/telemetry/index.js.map +0 -1
  163. package/dist/testing-C21CHsq2.d.ts +0 -20
  164. package/dist/testing.d.ts +0 -1
  165. package/dist/testing.js +0 -8
  166. package/dist/testing.js.map +0 -1
  167. package/dist/trajectory-2TkpSEVh.d.ts +0 -33
  168. package/dist/workflow/index.d.ts +0 -496
  169. package/dist/workflow/index.js +0 -2178
  170. package/dist/workflow/index.js.map +0 -1
  171. /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
  172. /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
  173. /package/dist/{chunk-OVPVM4JC.js.map → chunk-GTERJI6Q.js.map} +0 -0
  174. /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
  175. /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/index.js CHANGED
@@ -9,52 +9,34 @@ import {
9
9
  checkBehavioralCanary,
10
10
  checkCanaries,
11
11
  runBehavioralCanaries
12
- } from "./chunk-GDZAWO2I.js";
13
- import {
14
- classifyEuAiRisk,
15
- euAiActReport,
16
- nistAiRmfReport,
17
- renderMarkdown,
18
- soc2Report,
19
- summarize
20
- } from "./chunk-KKHDIONI.js";
21
- import {
22
- acquisitionPlansForKnowledgeGaps,
23
- blockingKnowledgeEval,
24
- knowledgeReadinessTracePayload,
25
- scoreKnowledgeReadiness,
26
- userQuestionsForKnowledgeGaps
27
- } from "./chunk-Q2JRAWRI.js";
28
- import {
29
- assertRecordIntegrity,
30
- checkRecordIntegrity,
31
- expandMatrix,
32
- gatePerf,
33
- scenarioKey,
34
- summarizeRecords
35
- } from "./chunk-STGVSCDH.js";
36
- import {
37
- assertProductBenchmarkRun,
38
- buildProductBenchmarkManifest,
39
- exportProductBenchmark,
40
- exportProductBenchmarkRuns,
41
- findProductBenchmarkArtifacts,
42
- productBenchmarkIntegrityFailures,
43
- productBenchmarkMutableSurfaces,
44
- productBenchmarkRepoIdentity,
45
- productBenchmarkSplits,
46
- readProductBenchmarkManifest,
47
- readProductBenchmarkRecords,
48
- runRecordToProductBenchmarkRecord,
49
- validateProductBenchmarkManifest,
50
- validateProductBenchmarkRecord,
51
- validateProductBenchmarkRun
52
- } from "./chunk-FOUG2VVS.js";
12
+ } from "./chunk-QFGTU7MT.js";
53
13
  import {
54
14
  BENCHMARK_SPLIT_SEED,
55
15
  benchmarks_exports,
56
16
  deterministicSplit
57
17
  } from "./chunk-T6W5ADLG.js";
18
+ import {
19
+ DEFAULT_RULES,
20
+ buildTrajectory,
21
+ classifyFailure,
22
+ compareToBaseline,
23
+ computeToolUseMetrics,
24
+ iqr,
25
+ welchsTTest
26
+ } from "./chunk-PLOMR3HP.js";
27
+ import {
28
+ analyzeSeries
29
+ } from "./chunk-BOD4O7OF.js";
30
+ import {
31
+ DockerSandboxDriver,
32
+ SandboxHarness,
33
+ SubprocessSandboxDriver,
34
+ composeParsers,
35
+ jestTestParser,
36
+ pytestTestParser,
37
+ runTestGradedScenario,
38
+ vitestTestParser
39
+ } from "./chunk-HZHNRYHK.js";
58
40
  import {
59
41
  CODING_HARNESSES,
60
42
  HARNESS_NATIVE_MODEL,
@@ -77,7 +59,7 @@ import {
77
59
  llmJudge,
78
60
  parseCorrectnessResponse,
79
61
  verifyCompletion
80
- } from "./chunk-6SKVFBTR.js";
62
+ } from "./chunk-RNB2NICW.js";
81
63
  import {
82
64
  DEFAULT_MUTATION_PRIMITIVES,
83
65
  DEFAULT_RED_TEAM_CORPUS,
@@ -101,17 +83,7 @@ import {
101
83
  scoreRedTeamOutput,
102
84
  surfaceContentHash,
103
85
  toolNamesForRun
104
- } from "./chunk-OVPVM4JC.js";
105
- import {
106
- BackendIntegrityError,
107
- assertRealBackend,
108
- cachedJudge,
109
- canonicalJson,
110
- contentHash,
111
- fileVerdictCache,
112
- inMemoryVerdictCache,
113
- summarizeBackendIntegrity
114
- } from "./chunk-3PFZBGMR.js";
86
+ } from "./chunk-GTERJI6Q.js";
115
87
  import {
116
88
  MODEL_PRICING,
117
89
  MetricsCollector,
@@ -122,33 +94,20 @@ import {
122
94
  resolveModelPricing
123
95
  } from "./chunk-VI2UW6B6.js";
124
96
  import {
125
- DEFAULT_RULES,
126
- classifyFailure,
127
- compareToBaseline,
128
- computeToolUseMetrics,
129
- iqr,
130
- welchsTTest
131
- } from "./chunk-DBDRR6GF.js";
132
- import {
133
- analyzeSeries
134
- } from "./chunk-BOD4O7OF.js";
135
- import {
136
- exportTrainingData,
137
- toNdjson
138
- } from "./chunk-KMPRBJK4.js";
139
- import {
140
- DockerSandboxDriver,
141
- SandboxHarness,
142
- SubprocessSandboxDriver,
143
- composeParsers,
144
- jestTestParser,
145
- pytestTestParser,
146
- runTestGradedScenario,
147
- vitestTestParser
148
- } from "./chunk-HZHNRYHK.js";
97
+ BackendIntegrityError,
98
+ assertRealBackend,
99
+ cachedJudge,
100
+ canonicalJson,
101
+ contentHash,
102
+ fileVerdictCache,
103
+ inMemoryVerdictCache,
104
+ summarizeBackendIntegrity
105
+ } from "./chunk-3PFZBGMR.js";
149
106
  import {
150
107
  DEFAULT_COMPLEXITY_WEIGHTS,
151
108
  FindingsStore,
109
+ LockedJsonlAppender,
110
+ Mutex,
152
111
  RunCritic,
153
112
  SEMANTIC_CONCEPT_JUDGE_VERSION,
154
113
  SKILL_USAGE_ANALYST,
@@ -159,15 +118,11 @@ import {
159
118
  defaultIsMaterial,
160
119
  diffFindings,
161
120
  runSemanticConceptJudge
162
- } from "./chunk-5PK3626Q.js";
163
- import {
164
- LockedJsonlAppender,
165
- Mutex
166
- } from "./chunk-DJWX3GVS.js";
121
+ } from "./chunk-XRGOKCMO.js";
167
122
  import {
168
123
  buildDefaultAnalystRegistry,
169
124
  computeTraceMetrics
170
- } from "./chunk-QRVS7MX4.js";
125
+ } from "./chunk-OW47B5WA.js";
171
126
  import {
172
127
  DEFAULT_RUN_SCORE_WEIGHTS,
173
128
  POLICY_EDIT_AXES,
@@ -184,7 +139,7 @@ import {
184
139
  policyEditsFromFindings,
185
140
  scorePolicyEditReadiness,
186
141
  validatePolicyEdit
187
- } from "./chunk-LOJ2QVCE.js";
142
+ } from "./chunk-2IY4ILP4.js";
188
143
  import {
189
144
  AnalystRegistry,
190
145
  DEFAULT_TRACE_ANALYST_KINDS,
@@ -192,32 +147,32 @@ import {
192
147
  IMPROVEMENT_KIND_SPEC,
193
148
  KNOWLEDGE_GAP_KIND_SPEC,
194
149
  KNOWLEDGE_POISONING_KIND_SPEC,
150
+ computeFindingId,
195
151
  createTraceAnalystKind,
152
+ makeFinding,
196
153
  renderPriorFindings
197
- } from "./chunk-2OGPXHOB.js";
154
+ } from "./chunk-7NX6ZSBG.js";
198
155
  import {
156
+ allCriticalPassed,
199
157
  controlFailureClassFromVerification,
200
158
  controlRunToRunRecord,
201
159
  createLlmReviewer,
160
+ errorStreakDetector,
202
161
  evaluateActionPolicy,
203
162
  inMemoryReviewStore,
204
163
  jsonlReviewStore,
205
- runProposeReview,
206
- runProposeReviewAsControlLoop,
207
- scoreFromEvals
208
- } from "./chunk-K7QEIHHJ.js";
209
- import {
210
- allCriticalPassed,
211
- errorStreakDetector,
212
164
  noProgressDetector,
213
165
  objectiveEval,
214
166
  observeAll,
215
167
  repeatedActionDetector,
216
168
  runAgentControlLoop,
169
+ runProposeReview,
170
+ runProposeReviewAsControlLoop,
171
+ scoreFromEvals,
217
172
  stopOnNoProgress,
218
173
  stopOnRepeatedAction,
219
174
  subjectiveEval
220
- } from "./chunk-YEHAEDUD.js";
175
+ } from "./chunk-IMWDSFUM.js";
221
176
  import {
222
177
  assertReleaseConfidence,
223
178
  bootstrapCi,
@@ -227,20 +182,8 @@ import {
227
182
  } from "./chunk-6HFYGEZ3.js";
228
183
  import {
229
184
  runEvalCampaign
230
- } from "./chunk-LIEJUH2I.js";
185
+ } from "./chunk-6PL5MGDL.js";
231
186
  import "./chunk-N22ZO7FV.js";
232
- import {
233
- LlmCallError,
234
- LlmClient,
235
- LlmRouteAssertionError,
236
- assertLlmRoute,
237
- backoffMs,
238
- callLlm,
239
- callLlmJson,
240
- isTransientLlmError,
241
- probeLlm,
242
- stripFencedJson
243
- } from "./chunk-FUCQVFMU.js";
244
187
  import {
245
188
  evaluateInterimReleaseConfidence,
246
189
  pairedEvalueSequence
@@ -252,17 +195,6 @@ import {
252
195
  researchReport,
253
196
  summaryTable
254
197
  } from "./chunk-6MFFKPZ4.js";
255
- import {
256
- attributeCounterfactuals,
257
- runCounterfactual
258
- } from "./chunk-6SK5VFYK.js";
259
- import {
260
- buildTrajectory
261
- } from "./chunk-RZTMDUO7.js";
262
- import {
263
- computeFindingId,
264
- makeFinding
265
- } from "./chunk-45EEMHTC.js";
266
198
  import {
267
199
  benjaminiHochberg,
268
200
  bonferroni,
@@ -331,7 +263,24 @@ import {
331
263
  scoreTraceInsightReadiness,
332
264
  tokenizeDomainWords,
333
265
  traceAnalystOnRunComplete
334
- } from "./chunk-V7HNA47Z.js";
266
+ } from "./chunk-RSVSSZKF.js";
267
+ import {
268
+ FAILURE_CLASSES,
269
+ TRACE_SCHEMA_VERSION,
270
+ aggregateLlm,
271
+ argHash,
272
+ groupBy,
273
+ isJudgeSpan,
274
+ isLlmSpan,
275
+ isRetrievalSpan,
276
+ isSandboxSpan,
277
+ isToolSpan,
278
+ judgeSpans,
279
+ llmSpans,
280
+ runFailureClass,
281
+ runsForScenario,
282
+ toolSpans
283
+ } from "./chunk-MHNQWM4I.js";
335
284
  import {
336
285
  TRACE_ANALYST_ACTOR_DESCRIPTION,
337
286
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
@@ -344,25 +293,6 @@ import {
344
293
  redactString,
345
294
  redactValue
346
295
  } from "./chunk-GGE4NNQT.js";
347
- import {
348
- aggregateLlm,
349
- argHash,
350
- groupBy,
351
- judgeSpans,
352
- llmSpans,
353
- runFailureClass,
354
- runsForScenario,
355
- toolSpans
356
- } from "./chunk-JZXGWLK5.js";
357
- import {
358
- FAILURE_CLASSES,
359
- TRACE_SCHEMA_VERSION,
360
- isJudgeSpan,
361
- isLlmSpan,
362
- isRetrievalSpan,
363
- isSandboxSpan,
364
- isToolSpan
365
- } from "./chunk-5BKGXME7.js";
366
296
  import {
367
297
  DEFAULT_TRACE_ANALYST_BUDGETS,
368
298
  LLM_CACHED_TOKENS,
@@ -403,12 +333,9 @@ import {
403
333
  throwIfRunIncomplete
404
334
  } from "./chunk-TT4KNT67.js";
405
335
  import {
406
- FileSystemRawProviderSink,
407
- InMemoryRawProviderSink,
408
- NoopRawProviderSink,
409
- defaultProviderRedactor,
410
- providerFromBaseUrl
411
- } from "./chunk-PC4UYEBM.js";
336
+ TraceEmitter,
337
+ llmSpanFromProvider
338
+ } from "./chunk-TVVP3ZZQ.js";
412
339
  import {
413
340
  RunRecordValidationError,
414
341
  isRunRecord,
@@ -417,10 +344,6 @@ import {
417
344
  roundTripRunRecord,
418
345
  validateRunRecord
419
346
  } from "./chunk-VK6HBGAE.js";
420
- import {
421
- TraceEmitter,
422
- llmSpanFromProvider
423
- } from "./chunk-TVVP3ZZQ.js";
424
347
  import {
425
348
  AGENT_PROFILE_KINDS,
426
349
  AgentProfileCellValidationError,
@@ -442,6 +365,25 @@ import {
442
365
  signManifest,
443
366
  verifyManifest
444
367
  } from "./chunk-VSMTAMNK.js";
368
+ import {
369
+ LlmCallError,
370
+ LlmClient,
371
+ LlmRouteAssertionError,
372
+ assertLlmRoute,
373
+ backoffMs,
374
+ callLlm,
375
+ callLlmJson,
376
+ isTransientLlmError,
377
+ probeLlm,
378
+ stripFencedJson
379
+ } from "./chunk-FUCQVFMU.js";
380
+ import {
381
+ FileSystemRawProviderSink,
382
+ InMemoryRawProviderSink,
383
+ NoopRawProviderSink,
384
+ defaultProviderRedactor,
385
+ providerFromBaseUrl
386
+ } from "./chunk-PC4UYEBM.js";
445
387
  import {
446
388
  AgentEvalError,
447
389
  CaptureIntegrityError,
@@ -654,12 +596,12 @@ function ghCliClient(opts = {}) {
654
596
  await exec("git", ["branch", "-D", input.branchName], { cwd });
655
597
  await run("git", ["checkout", "-b", input.branchName]);
656
598
  const { mkdir, writeFile } = await import("fs/promises");
657
- const { dirname: dirname3, join: join4, resolve } = await import("path");
599
+ const { dirname: dirname5, join: join6, resolve: resolve2 } = await import("path");
658
600
  for (const change of input.fileChanges) {
659
- const abs = resolve(cwd, change.path);
660
- await mkdir(dirname3(abs), { recursive: true });
601
+ const abs = resolve2(cwd, change.path);
602
+ await mkdir(dirname5(abs), { recursive: true });
661
603
  await writeFile(abs, change.contents, "utf8");
662
- await run("git", ["add", join4(change.path)]);
604
+ await run("git", ["add", join6(change.path)]);
663
605
  }
664
606
  const env = {};
665
607
  if (input.authorName) env.GIT_AUTHOR_NAME = input.authorName;
@@ -1121,7 +1063,7 @@ function capabilityHeadroom(rows, opts = {}) {
1121
1063
  tasksWithGap: tasks.filter((t) => t.headroom === "gap").length,
1122
1064
  tasksSaturated: tasks.filter((t) => t.headroom === "saturated").length,
1123
1065
  tasksUnknown: tasks.filter((t) => t.headroom === "unknown").length,
1124
- repsUnknown: tasks.reduce((sum3, t) => sum3 + (t.n - t.nKnown), 0)
1066
+ repsUnknown: tasks.reduce((sum4, t) => sum4 + (t.n - t.nKnown), 0)
1125
1067
  };
1126
1068
  return { tasks, summary };
1127
1069
  }
@@ -1711,10 +1653,10 @@ var FileSystemFeedbackTrajectoryStore = class {
1711
1653
  }
1712
1654
  async append(record) {
1713
1655
  const { appendFile, mkdir } = await import("fs/promises");
1714
- const { join: join4 } = await import("path");
1656
+ const { join: join6 } = await import("path");
1715
1657
  await mkdir(this.dir, { recursive: true });
1716
1658
  await appendFile(
1717
- join4(this.dir, "feedback-trajectories.ndjson"),
1659
+ join6(this.dir, "feedback-trajectories.ndjson"),
1718
1660
  `${JSON.stringify(record)}
1719
1661
  `,
1720
1662
  "utf8"
@@ -1723,8 +1665,8 @@ var FileSystemFeedbackTrajectoryStore = class {
1723
1665
  async load() {
1724
1666
  if (this.loaded) return;
1725
1667
  const { readFile: readFile2 } = await import("fs/promises");
1726
- const { join: join4 } = await import("path");
1727
- const file = join4(this.dir, "feedback-trajectories.ndjson");
1668
+ const { join: join6 } = await import("path");
1669
+ const file = join6(this.dir, "feedback-trajectories.ndjson");
1728
1670
  try {
1729
1671
  const raw = await readFile2(file, "utf8");
1730
1672
  for (const line of raw.split("\n")) {
@@ -1971,7 +1913,7 @@ function scoreFromLabels(labels) {
1971
1913
  return void 0;
1972
1914
  }).filter((value) => typeof value === "number");
1973
1915
  if (!scored.length) return void 0;
1974
- return Math.round(scored.reduce((sum3, value) => sum3 + value, 0) / scored.length * 1e3) / 1e3;
1916
+ return Math.round(scored.reduce((sum4, value) => sum4 + value, 0) / scored.length * 1e3) / 1e3;
1975
1917
  }
1976
1918
  function instructionFromLabel(trajectory, label) {
1977
1919
  if (label.kind === "reject" && label.reason)
@@ -2260,6 +2202,190 @@ function assertCrossFamily(models, opts = {}) {
2260
2202
  return list;
2261
2203
  }
2262
2204
 
2205
+ // src/knowledge/readiness.ts
2206
+ function scoreKnowledgeReadiness(options) {
2207
+ const now = options.now ?? /* @__PURE__ */ new Date();
2208
+ const requirements = options.requirements.map(normalizeRequirement);
2209
+ const missing = requirements.filter((requirement) => isRequirementMissing(requirement, now));
2210
+ const blockingMissingRequirements = missing.filter(isBlockingGap);
2211
+ const nonBlockingGaps = missing.filter((requirement) => !isBlockingGap(requirement));
2212
+ const readinessScore = weightedReadinessAt(requirements, now);
2213
+ const bundle = {
2214
+ taskId: options.taskId,
2215
+ requirements,
2216
+ evidenceIds: unique([
2217
+ ...options.evidenceIds ?? [],
2218
+ ...requirements.flatMap((r) => r.evidenceIds)
2219
+ ]),
2220
+ claimIds: unique(options.claimIds ?? []),
2221
+ wikiPageIds: unique(options.wikiPageIds ?? []),
2222
+ userAnswers: options.userAnswers ?? {},
2223
+ missing,
2224
+ readinessScore,
2225
+ metadata: options.metadata
2226
+ };
2227
+ const recommendedAction = chooseRecommendedAction(blockingMissingRequirements, nonBlockingGaps);
2228
+ const severity = blockingMissingRequirements.length > 0 ? "critical" : nonBlockingGaps.some((gap) => gap.importance === "high") ? "warning" : "info";
2229
+ const reason = blockingMissingRequirements.length > 0 ? `${blockingMissingRequirements.length} blocking knowledge requirement(s) are missing.` : nonBlockingGaps.length > 0 ? `${nonBlockingGaps.length} non-blocking knowledge gap(s) remain.` : "All declared knowledge requirements are ready.";
2230
+ return {
2231
+ taskId: options.taskId,
2232
+ readinessScore,
2233
+ blockingMissingRequirements,
2234
+ nonBlockingGaps,
2235
+ recommendedAction,
2236
+ bundle,
2237
+ severity,
2238
+ reason
2239
+ };
2240
+ }
2241
+ function blockingKnowledgeEval(report, options = {}) {
2242
+ const minimumScore = options.minimumScore ?? 0.7;
2243
+ const passed = report.blockingMissingRequirements.length === 0 && report.readinessScore >= minimumScore;
2244
+ if (options.emitter) {
2245
+ void options.emitter.emit({
2246
+ kind: "custom",
2247
+ payload: knowledgeReadinessTracePayload(report, { passed, minimumScore })
2248
+ }).catch(() => void 0);
2249
+ }
2250
+ return objectiveEval({
2251
+ id: options.id ?? "knowledge-ready",
2252
+ passed,
2253
+ score: report.readinessScore,
2254
+ severity: passed ? "info" : report.severity,
2255
+ detail: report.reason,
2256
+ evidence: report.blockingMissingRequirements.map((r) => r.id).join(", ") || void 0,
2257
+ metadata: { knowledgeReadiness: report }
2258
+ });
2259
+ }
2260
+ function knowledgeReadinessTracePayload(report, options = {}) {
2261
+ return {
2262
+ kind: "readiness_scored",
2263
+ taskId: report.taskId,
2264
+ passed: options.passed ?? report.blockingMissingRequirements.length === 0,
2265
+ readinessScore: report.readinessScore,
2266
+ minimumScore: options.minimumScore,
2267
+ blockingRequirementIds: report.blockingMissingRequirements.map((r) => r.id),
2268
+ nonBlockingRequirementIds: report.nonBlockingGaps.map((r) => r.id),
2269
+ recommendedAction: report.recommendedAction,
2270
+ severity: report.severity,
2271
+ reason: report.reason
2272
+ };
2273
+ }
2274
+ function userQuestionsForKnowledgeGaps(gaps) {
2275
+ return gaps.filter((gap) => gap.acquisitionMode === "ask_user" || gap.fallbackPolicy === "ask").map((gap) => ({
2276
+ id: `question_${gap.id}`,
2277
+ question: `Please provide: ${gap.description}`,
2278
+ reason: `Required for ${gap.requiredFor.join(", ") || "the task"}.`,
2279
+ requirementId: gap.id,
2280
+ importance: gap.importance,
2281
+ answerType: gap.sensitivity === "secret" ? "credential" : "free_text",
2282
+ impactIfUnknown: impactFor(gap)
2283
+ }));
2284
+ }
2285
+ function acquisitionPlansForKnowledgeGaps(gaps) {
2286
+ const byMode = /* @__PURE__ */ new Map();
2287
+ for (const gap of gaps) {
2288
+ const mode = planMode(gap.acquisitionMode);
2289
+ if (!mode) continue;
2290
+ const bucket = byMode.get(mode) ?? [];
2291
+ bucket.push(gap);
2292
+ byMode.set(mode, bucket);
2293
+ }
2294
+ return [...byMode.entries()].map(([mode, requirements]) => ({
2295
+ id: `acquire_${mode}`,
2296
+ requirementIds: requirements.map((r) => r.id),
2297
+ mode,
2298
+ description: descriptionForPlan(mode, requirements),
2299
+ priority: maxImportance(requirements.map((r) => r.importance)),
2300
+ questions: mode === "ask_user" ? userQuestionsForKnowledgeGaps(requirements) : void 0
2301
+ }));
2302
+ }
2303
+ function normalizeRequirement(requirement) {
2304
+ return {
2305
+ ...requirement,
2306
+ confidenceNeeded: clamp012(requirement.confidenceNeeded),
2307
+ currentConfidence: clamp012(requirement.currentConfidence),
2308
+ evidenceIds: unique(requirement.evidenceIds)
2309
+ };
2310
+ }
2311
+ function weightedReadinessAt(requirements, now) {
2312
+ if (requirements.length === 0) return 1;
2313
+ let weightSum = 0;
2314
+ let scoreSum = 0;
2315
+ for (const requirement of requirements) {
2316
+ const weight = importanceWeight(requirement.importance);
2317
+ const score = isExpired(requirement, now) ? 0 : requirement.confidenceNeeded <= 0 ? 1 : Math.min(1, requirement.currentConfidence / requirement.confidenceNeeded);
2318
+ weightSum += weight;
2319
+ scoreSum += weight * score;
2320
+ }
2321
+ return clamp012(scoreSum / weightSum);
2322
+ }
2323
+ function isRequirementMissing(requirement, now) {
2324
+ return isExpired(requirement, now) || requirement.currentConfidence < requirement.confidenceNeeded;
2325
+ }
2326
+ function isExpired(requirement, now) {
2327
+ if (!requirement.validUntil) return false;
2328
+ const deadline = Date.parse(requirement.validUntil);
2329
+ if (!Number.isFinite(deadline)) return true;
2330
+ return deadline <= now.getTime();
2331
+ }
2332
+ function isBlockingGap(requirement) {
2333
+ return requirement.importance === "blocking" || requirement.fallbackPolicy === "block" || requirement.sensitivity === "secret";
2334
+ }
2335
+ function chooseRecommendedAction(blocking, nonBlocking) {
2336
+ const gaps = blocking.length > 0 ? blocking : nonBlocking;
2337
+ if (gaps.length === 0) return "run_agent";
2338
+ if (gaps.some((gap) => gap.acquisitionMode === "ask_user" || gap.fallbackPolicy === "ask"))
2339
+ return "ask_user";
2340
+ if (gaps.some((gap) => gap.acquisitionMode === "query_connector")) return "query_connectors";
2341
+ if (gaps.some(
2342
+ (gap) => gap.acquisitionMode === "inspect_repo" || gap.acquisitionMode === "run_command"
2343
+ ))
2344
+ return "inspect_repo";
2345
+ if (gaps.some((gap) => gap.acquisitionMode === "search_web")) return "collect_web_data";
2346
+ if (gaps.some((gap) => gap.acquisitionMode === "not_available")) return "abort_or_rescope";
2347
+ if (nonBlocking.some((gap) => gap.importance === "high")) return "build_domain_wiki";
2348
+ return "continue_with_caveat";
2349
+ }
2350
+ function planMode(mode) {
2351
+ if (mode === "infer_low_confidence" || mode === "not_available") return null;
2352
+ return mode;
2353
+ }
2354
+ function descriptionForPlan(mode, requirements) {
2355
+ const labels = requirements.map((r) => r.description).join("; ");
2356
+ if (mode === "ask_user") return `Ask the user for: ${labels}`;
2357
+ if (mode === "search_web") return `Search web or documentation sources for: ${labels}`;
2358
+ if (mode === "query_connector") return `Query configured connectors for: ${labels}`;
2359
+ if (mode === "inspect_repo") return `Inspect repository context for: ${labels}`;
2360
+ if (mode === "run_command") return `Run local commands to collect: ${labels}`;
2361
+ return `Build domain wiki evidence for: ${labels}`;
2362
+ }
2363
+ function impactFor(requirement) {
2364
+ if (requirement.fallbackPolicy === "block") return "The agent should not run until this is known.";
2365
+ if (requirement.fallbackPolicy === "continue_with_caveat")
2366
+ return "The agent may continue, but must disclose uncertainty.";
2367
+ if (requirement.fallbackPolicy === "use_default")
2368
+ return "The agent will use the configured default if skipped.";
2369
+ return "The agent should ask before continuing.";
2370
+ }
2371
+ function maxImportance(values) {
2372
+ const order = ["blocking", "high", "medium", "low"];
2373
+ return order.find((value) => values.includes(value)) ?? "low";
2374
+ }
2375
+ function importanceWeight(importance) {
2376
+ if (importance === "blocking") return 8;
2377
+ if (importance === "high") return 4;
2378
+ if (importance === "medium") return 2;
2379
+ return 1;
2380
+ }
2381
+ function clamp012(value) {
2382
+ if (!Number.isFinite(value)) return 0;
2383
+ return Math.max(0, Math.min(1, value));
2384
+ }
2385
+ function unique(items) {
2386
+ return [...new Set(items)];
2387
+ }
2388
+
2263
2389
  // src/live-proof.ts
2264
2390
  async function runLiveProof(config) {
2265
2391
  const startedAt = Date.now();
@@ -3520,7 +3646,7 @@ async function mapLimit(items, limit, fn) {
3520
3646
  return results;
3521
3647
  }
3522
3648
  function mean2(values) {
3523
- return values.length ? values.reduce((sum3, value) => sum3 + value, 0) / values.length : 0;
3649
+ return values.length ? values.reduce((sum4, value) => sum4 + value, 0) / values.length : 0;
3524
3650
  }
3525
3651
  function meanRunScore(scores2) {
3526
3652
  return {
@@ -3612,7 +3738,7 @@ function aggregateJudgeVerdicts(verdicts, dimensionKeys, weights) {
3612
3738
  var DEFAULT_MAX_ATTEMPTS = 3;
3613
3739
  var DEFAULT_TIMEOUT_MS = 3e5;
3614
3740
  function sleep(ms) {
3615
- return new Promise((resolve) => setTimeout(resolve, ms));
3741
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
3616
3742
  }
3617
3743
  async function withJudgeRetry(judgeFn, policy = {}) {
3618
3744
  const maxAttempts = policy.maxAttempts ?? DEFAULT_MAX_ATTEMPTS;
@@ -4041,7 +4167,7 @@ function rankRows(rows, weights) {
4041
4167
  }
4042
4168
  return [...buckets.entries()].map(([variantId, values]) => ({
4043
4169
  variantId,
4044
- mean: values.reduce((sum3, value) => sum3 + value, 0) / values.length,
4170
+ mean: values.reduce((sum4, value) => sum4 + value, 0) / values.length,
4045
4171
  runs: values.length
4046
4172
  })).sort((a, b) => b.mean - a.mean);
4047
4173
  }
@@ -4295,13 +4421,13 @@ var defaultBlendWeights = { heldout: 0.7, judge: 0.3 };
4295
4421
  function normalizeWeights(weights) {
4296
4422
  const h = Number.isFinite(weights.heldout) && weights.heldout >= 0 ? weights.heldout : 0;
4297
4423
  const j = Number.isFinite(weights.judge) && weights.judge >= 0 ? weights.judge : 0;
4298
- const sum3 = h + j;
4299
- if (sum3 <= 0) {
4424
+ const sum4 = h + j;
4425
+ if (sum4 <= 0) {
4300
4426
  throw new ValidationError(
4301
4427
  "blend weights must have a positive sum (got heldout+judge <= 0) \u2014 cannot weight a composite by zero"
4302
4428
  );
4303
4429
  }
4304
- return { heldout: h / sum3, judge: j / sum3 };
4430
+ return { heldout: h / sum4, judge: j / sum4 };
4305
4431
  }
4306
4432
  function blendHeldout(heldoutPassRate, judgeScore, weights = defaultBlendWeights) {
4307
4433
  const w = normalizeWeights(weights);
@@ -6705,7 +6831,7 @@ async function commitBisect(options) {
6705
6831
  }
6706
6832
  async function promptBisect(options) {
6707
6833
  const split = options.paragraphSplitter ?? ((p) => p.split(/\n\s*\n/));
6708
- const join4 = (paragraphs) => paragraphs.join("\n\n");
6834
+ const join6 = (paragraphs) => paragraphs.join("\n\n");
6709
6835
  const goodParas = split(options.good);
6710
6836
  const badParas = split(options.bad);
6711
6837
  if (goodParas.length !== badParas.length) {
@@ -6725,7 +6851,7 @@ async function promptBisect(options) {
6725
6851
  const result = await bisect({
6726
6852
  good: goodMask,
6727
6853
  bad: badMask,
6728
- runEval: (mask) => options.runEval(join4(paragraphsFor(mask))),
6854
+ runEval: (mask) => options.runEval(join6(paragraphsFor(mask))),
6729
6855
  maxIterations: options.maxIterations ?? n + 5,
6730
6856
  halfway: (g, b) => {
6731
6857
  for (let i = 0; i < g.length; i++) {
@@ -6756,12 +6882,12 @@ async function promptBisect(options) {
6756
6882
  }
6757
6883
  }
6758
6884
  const materializedPath = result.path.map((s) => ({
6759
- state: join4(paragraphsFor(s.state)),
6885
+ state: join6(paragraphsFor(s.state)),
6760
6886
  score: s.score,
6761
6887
  pass: s.pass
6762
6888
  }));
6763
6889
  return {
6764
- culprit: join4(paragraphsFor(culprit)),
6890
+ culprit: join6(paragraphsFor(culprit)),
6765
6891
  path: materializedPath,
6766
6892
  converged: result.converged,
6767
6893
  inputInconsistent: result.inputInconsistent,
@@ -6769,6 +6895,90 @@ async function promptBisect(options) {
6769
6895
  };
6770
6896
  }
6771
6897
 
6898
+ // src/counterfactual.ts
6899
+ async function runCounterfactual(store, originalRunId, mutation, runner) {
6900
+ const originalRun = await store.getRun(originalRunId);
6901
+ if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
6902
+ const trajectory = await buildTrajectory(store, originalRunId);
6903
+ if (mutation.at < 0 || mutation.at >= trajectory.steps.length) {
6904
+ throw new ValidationError(
6905
+ `counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`
6906
+ );
6907
+ }
6908
+ const targetStep = trajectory.steps[mutation.at];
6909
+ const mutatedStep = applyMutation(targetStep, mutation);
6910
+ const cfEmitter = new TraceEmitter(store);
6911
+ await cfEmitter.startRun({
6912
+ scenarioId: originalRun.scenarioId,
6913
+ variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
6914
+ projectId: originalRun.projectId,
6915
+ parentRunId: originalRunId,
6916
+ layer: "meta",
6917
+ tags: { counterfactual: "true", mutationKind: mutation.kind, mutationAt: String(mutation.at) }
6918
+ });
6919
+ await runner.executeFrom(
6920
+ {
6921
+ originalRunId,
6922
+ originalTrajectory: trajectory,
6923
+ prefix: trajectory.steps.slice(0, mutation.at),
6924
+ mutation,
6925
+ mutatedStep
6926
+ },
6927
+ cfEmitter
6928
+ );
6929
+ const counterfactual = await store.getRun(cfEmitter.runId);
6930
+ const delta = {
6931
+ originalOutcomeScore: originalRun.outcome?.score ?? null,
6932
+ counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
6933
+ deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
6934
+ };
6935
+ return { counterfactualRunId: cfEmitter.runId, originalRunId, mutation, delta };
6936
+ }
6937
+ function applyMutation(step, mutation) {
6938
+ if (mutation.kind === "swap-model" && step.span.kind === "llm") {
6939
+ const llm = step.span;
6940
+ return { ...step, span: { ...llm, model: mutation.newModel } };
6941
+ }
6942
+ if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
6943
+ const tool = step.span;
6944
+ return { ...step, span: { ...tool, result: mutation.newResult } };
6945
+ }
6946
+ if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
6947
+ const llm = step.span;
6948
+ return {
6949
+ ...step,
6950
+ span: {
6951
+ ...llm,
6952
+ messages: [{ role: "system", content: mutation.content }, ...llm.messages]
6953
+ }
6954
+ };
6955
+ }
6956
+ if (mutation.kind === "custom") return mutation.apply(step);
6957
+ return step;
6958
+ }
6959
+ function attributeCounterfactuals(results) {
6960
+ const grouped = /* @__PURE__ */ new Map();
6961
+ for (const r of results) {
6962
+ const arr = grouped.get(r.mutation.kind) ?? [];
6963
+ arr.push(r);
6964
+ grouped.set(r.mutation.kind, arr);
6965
+ }
6966
+ const out = [];
6967
+ for (const [kind, items] of grouped) {
6968
+ const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
6969
+ if (deltas.length === 0) continue;
6970
+ const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
6971
+ const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
6972
+ out.push({
6973
+ mutationKind: kind,
6974
+ n: deltas.length,
6975
+ meanAbsDelta: meanAbs,
6976
+ meanSignedDelta: meanSigned
6977
+ });
6978
+ }
6979
+ return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
6980
+ }
6981
+
6772
6982
  // src/cross-trace-diff.ts
6773
6983
  async function crossTraceDiff(store, runA, runB, options = {}) {
6774
6984
  const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
@@ -7051,6 +7261,67 @@ function groupBy2(items, key) {
7051
7261
  return m;
7052
7262
  }
7053
7263
 
7264
+ // src/prm/training-export.ts
7265
+ async function exportTrainingData(store, graded, options = {}) {
7266
+ const window = options.contextWindow ?? 5;
7267
+ const out = [];
7268
+ for (const g of graded) {
7269
+ const trajectory = await buildTrajectory(store, g.runId);
7270
+ const spanById = new Map(trajectory.steps.map((s) => [s.span.spanId, s]));
7271
+ for (const gs of g.steps) {
7272
+ const node = spanById.get(gs.spanId);
7273
+ if (!node) continue;
7274
+ const idx = trajectory.steps.indexOf(node);
7275
+ const priorSpans = trajectory.steps.slice(Math.max(0, idx - window), idx).map((s) => s.span);
7276
+ out.push({
7277
+ runId: g.runId,
7278
+ spanId: gs.spanId,
7279
+ rubricId: gs.rubricId,
7280
+ score: gs.score,
7281
+ context: {
7282
+ priorTurns: priorSpans.map(spanToTurn).filter((t) => t !== null),
7283
+ step: { kind: node.span.kind, text: spanToText(node.span) }
7284
+ },
7285
+ rationale: gs.rationale,
7286
+ evidence: gs.evidence
7287
+ });
7288
+ }
7289
+ }
7290
+ return out;
7291
+ }
7292
+ function toNdjson(samples) {
7293
+ return `${samples.map((s) => JSON.stringify(s)).join("\n")}
7294
+ `;
7295
+ }
7296
+ function spanToTurn(span) {
7297
+ if (isLlmSpan(span)) {
7298
+ const text = span.output ?? span.messages.map((m) => `${m.role}: ${m.content}`).join("\n");
7299
+ return { role: "assistant", content: text };
7300
+ }
7301
+ if (isToolSpan(span)) {
7302
+ return {
7303
+ role: "tool",
7304
+ content: `${span.toolName}(${safeStringify(span.args)}) \u2192 ${safeStringify(span.result)}`
7305
+ };
7306
+ }
7307
+ return null;
7308
+ }
7309
+ function spanToText(span) {
7310
+ if (isLlmSpan(span)) return span.output ?? "";
7311
+ if (isToolSpan(span))
7312
+ return `${span.toolName}(${safeStringify(span.args)}) \u2192 ${safeStringify(span.result)}`;
7313
+ return span.name;
7314
+ }
7315
+ function safeStringify(v) {
7316
+ if (v === null || v === void 0) return "";
7317
+ if (typeof v === "string") return v;
7318
+ try {
7319
+ return JSON.stringify(v);
7320
+ } catch {
7321
+ return String(v);
7322
+ }
7323
+ }
7324
+
7054
7325
  // src/reward-model-export.ts
7055
7326
  async function exportRewardModel(store, grader, runIds) {
7056
7327
  const graded = await Promise.all(runIds.map((id) => grader.grade(store, id)));
@@ -7481,7 +7752,7 @@ function extractErrorCount(text, opts = {}) {
7481
7752
  for (const p of patterns) {
7482
7753
  const matches2 = Array.from(text.matchAll(p.regex));
7483
7754
  if (matches2.length === 0) continue;
7484
- const count = p.transform ? matches2.reduce((sum3, m) => sum3 + p.transform(m), 0) : matches2.length;
7755
+ const count = p.transform ? matches2.reduce((sum4, m) => sum4 + p.transform(m), 0) : matches2.length;
7485
7756
  return {
7486
7757
  count,
7487
7758
  matched: p.name,
@@ -8419,7 +8690,7 @@ function defaultReferenceReplayMatcher(reference, candidate) {
8419
8690
  const textScore = tokenJaccard(referenceText, candidateText);
8420
8691
  const severityScore = reference.severity && candidate.severity ? normalize(reference.severity) === normalize(candidate.severity) ? 0.1 : -0.05 : 0;
8421
8692
  const tagScore = tagOverlap(reference.tags, candidate.tags) * 0.15;
8422
- const score = clamp012(textScore * 0.85 + tagScore + severityScore);
8693
+ const score = clamp013(textScore * 0.85 + tagScore + severityScore);
8423
8694
  return {
8424
8695
  score,
8425
8696
  reason: `token=${textScore.toFixed(2)} tags=${tagScore.toFixed(2)} severity=${severityScore.toFixed(2)}`
@@ -8525,13 +8796,13 @@ function scorePair(scenario, matcher, reference, candidate) {
8525
8796
  `reference replay matcher returned non-finite score for ${scenario.id}:${reference.id}:${candidate.id}`
8526
8797
  );
8527
8798
  }
8528
- return { score: clamp012(result.score), reason: result.reason ?? "" };
8799
+ return { score: clamp013(result.score), reason: result.reason ?? "" };
8529
8800
  }
8530
8801
  function buildScenarioScore(scenario, matches2, falsePositives) {
8531
8802
  const matched = matches2.filter((match) => match.matched).length;
8532
8803
  const total = scenario.references.length;
8533
- const matchedWeight = matches2.filter((match) => match.matched).reduce((sum3, match) => sum3 + match.weight, 0);
8534
- const totalWeight = matches2.reduce((sum3, match) => sum3 + match.weight, 0);
8804
+ const matchedWeight = matches2.filter((match) => match.matched).reduce((sum4, match) => sum4 + match.weight, 0);
8805
+ const totalWeight = matches2.reduce((sum4, match) => sum4 + match.weight, 0);
8535
8806
  const precision2 = ratio(matched, matched + falsePositives);
8536
8807
  const recall = ratio(matched, total);
8537
8808
  return {
@@ -8624,7 +8895,7 @@ function tokens(text) {
8624
8895
  function normalize(text) {
8625
8896
  return text.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
8626
8897
  }
8627
- function clamp012(value) {
8898
+ function clamp013(value) {
8628
8899
  if (!Number.isFinite(value)) return 0;
8629
8900
  return Math.max(0, Math.min(1, value));
8630
8901
  }
@@ -9428,8 +9699,8 @@ function createSandboxPool(opts) {
9428
9699
  });
9429
9700
  return state;
9430
9701
  }
9431
- return new Promise((resolve, reject) => {
9432
- waiters.push({ resolve, reject });
9702
+ return new Promise((resolve2, reject) => {
9703
+ waiters.push({ resolve: resolve2, reject });
9433
9704
  });
9434
9705
  }
9435
9706
  function handOffCleanSlot(state) {
@@ -9617,7 +9888,7 @@ function traceJudge(judge, judgeName, opts) {
9617
9888
  });
9618
9889
  try {
9619
9890
  const scores2 = await judge(tc, input);
9620
- const composite = scores2.length > 0 ? scores2.reduce((sum3, s) => sum3 + s.score, 0) / scores2.length : 0;
9891
+ const composite = scores2.length > 0 ? scores2.reduce((sum4, s) => sum4 + s.score, 0) / scores2.length : 0;
9621
9892
  await span.end({
9622
9893
  attributes: {
9623
9894
  "judge.name": judgeName,
@@ -9662,7 +9933,7 @@ function traceJudgeEnsemble(judges, judgeNames, opts) {
9662
9933
  failedJudges++;
9663
9934
  }
9664
9935
  }
9665
- const composite = allScores.length > 0 ? allScores.reduce((sum3, s) => sum3 + s.score, 0) / allScores.length : 0;
9936
+ const composite = allScores.length > 0 ? allScores.reduce((sum4, s) => sum4 + s.score, 0) / allScores.length : 0;
9666
9937
  await ensembleSpan.end({
9667
9938
  attributes: {
9668
9939
  "judge.ensemble_size": judges.length,
@@ -10119,7 +10390,7 @@ function costReport(ledger) {
10119
10390
  perChannel: summary.byChannel,
10120
10391
  total: {
10121
10392
  usd: summary.totalCostUsd,
10122
- unknownEntries: summary.byChannel.reduce((sum3, c) => sum3 + c.unpricedCalls, 0)
10393
+ unknownEntries: summary.byChannel.reduce((sum4, c) => sum4 + c.unpricedCalls, 0)
10123
10394
  },
10124
10395
  perModel: [...perModel.values()].sort((a, b) => a.model.localeCompare(b.model))
10125
10396
  };
@@ -10217,6 +10488,839 @@ function verifyAttestation(report, attested) {
10217
10488
  }
10218
10489
  return { valid: true };
10219
10490
  }
10491
+
10492
+ // src/product-benchmark/index.ts
10493
+ import { existsSync as existsSync6, readFileSync as readFileSync7, statSync as statSync3 } from "fs";
10494
+ import { dirname as dirname4, join as join5 } from "path";
10495
+
10496
+ // src/product-benchmark/export.ts
10497
+ import { spawnSync as spawnSync2 } from "child_process";
10498
+ import { createHash } from "crypto";
10499
+ import { cpSync, existsSync as existsSync5, mkdirSync as mkdirSync3, readFileSync as readFileSync6, writeFileSync } from "fs";
10500
+ import { basename as basename2, dirname as dirname3, isAbsolute, join as join4, relative, resolve } from "path";
10501
+ var productBenchmarkMutableSurfaces = [
10502
+ "prompt",
10503
+ "resources.files",
10504
+ "tools",
10505
+ "mcp",
10506
+ "hooks",
10507
+ "subagents"
10508
+ ];
10509
+ function sha256(value) {
10510
+ return `sha256:${createHash("sha256").update(value).digest("hex")}`;
10511
+ }
10512
+ function safePathPart(value) {
10513
+ return value.replace(/[^a-zA-Z0-9_.-]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 80) || "artifact";
10514
+ }
10515
+ function git(args, fallback) {
10516
+ const res = spawnSync2("git", args, { cwd: process.cwd(), encoding: "utf8" });
10517
+ const value = res.status === 0 ? res.stdout.trim() : "";
10518
+ return value.length > 0 ? value : fallback;
10519
+ }
10520
+ function knownEnv(name) {
10521
+ const value = process.env[name]?.trim();
10522
+ return value && value !== "unknown" ? value : void 0;
10523
+ }
10524
+ function ciBranchName() {
10525
+ const direct = knownEnv("GITHUB_HEAD_REF") ?? knownEnv("GITHUB_REF_NAME") ?? knownEnv("VERCEL_GIT_COMMIT_REF");
10526
+ if (direct) return direct;
10527
+ const ref = knownEnv("GITHUB_REF");
10528
+ return ref?.startsWith("refs/heads/") ? ref.slice("refs/heads/".length) : void 0;
10529
+ }
10530
+ function repoBranch() {
10531
+ const ci = ciBranchName();
10532
+ if (ci) return ci;
10533
+ const branch = git(["branch", "--show-current"], "");
10534
+ if (branch) return branch;
10535
+ const name = git(["name-rev", "--name-only", "--exclude=tags/*", "HEAD"], "");
10536
+ if (name && name !== "undefined") return name;
10537
+ return `detached:${git(["rev-parse", "--short", "HEAD"], "unknown")}`;
10538
+ }
10539
+ function productBenchmarkRepoIdentity() {
10540
+ return {
10541
+ url: git(["config", "--get", "remote.origin.url"], "unknown"),
10542
+ commit: git(["rev-parse", "HEAD"], "unknown"),
10543
+ branch: repoBranch()
10544
+ };
10545
+ }
10546
+ function packageVersion(name) {
10547
+ const pkg = JSON.parse(readFileSync6(resolve("package.json"), "utf8"));
10548
+ if (pkg.name === name && pkg.version) return pkg.version;
10549
+ const declared = pkg.dependencies?.[name] ?? pkg.devDependencies?.[name];
10550
+ if (declared) return declared;
10551
+ const installed = resolve("node_modules", name, "package.json");
10552
+ if (existsSync5(installed)) {
10553
+ const installedPkg = JSON.parse(readFileSync6(installed, "utf8"));
10554
+ if (installedPkg.version) return installedPkg.version;
10555
+ }
10556
+ return "unknown";
10557
+ }
10558
+ function readRunRecords(path) {
10559
+ const lines = readFileSync6(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
10560
+ return lines.map((line, index) => {
10561
+ let parsed;
10562
+ try {
10563
+ parsed = JSON.parse(line);
10564
+ } catch (err) {
10565
+ throw new ValidationError(
10566
+ `${path}:${index + 1}: ${err instanceof Error ? err.message : String(err)}`,
10567
+ { cause: err }
10568
+ );
10569
+ }
10570
+ return expectRunRecordShape(parsed, `${path}:${index + 1}`);
10571
+ });
10572
+ }
10573
+ function expectRunRecordShape(value, path) {
10574
+ if (!value || typeof value !== "object" || Array.isArray(value)) {
10575
+ throw new ValidationError(`${path}: run record must be an object`);
10576
+ }
10577
+ const obj = value;
10578
+ for (const key of ["runId", "model"]) {
10579
+ if (typeof obj[key] !== "string" || obj[key].length === 0) {
10580
+ throw new ValidationError(`${path}: run record ${key} must be a non-empty string`);
10581
+ }
10582
+ }
10583
+ for (const key of ["costUsd", "wallMs"]) {
10584
+ if (typeof obj[key] !== "number" || !Number.isFinite(obj[key])) {
10585
+ throw new ValidationError(`${path}: run record ${key} must be a finite number`);
10586
+ }
10587
+ }
10588
+ const tokenUsage = obj.tokenUsage;
10589
+ if (!tokenUsage || typeof tokenUsage.input !== "number" || typeof tokenUsage.output !== "number") {
10590
+ throw new ValidationError(`${path}: run record tokenUsage.input/output must be numbers`);
10591
+ }
10592
+ const outcome = obj.outcome;
10593
+ if (!outcome || !outcome.raw || typeof outcome.raw !== "object") {
10594
+ throw new ValidationError(`${path}: run record outcome.raw must be an object`);
10595
+ }
10596
+ return value;
10597
+ }
10598
+ function clamp014(value) {
10599
+ return Math.max(0, Math.min(1, value));
10600
+ }
10601
+ function splitOf(record, opts) {
10602
+ const custom = opts.classifySplit?.(record);
10603
+ if (custom) return custom;
10604
+ if (record.outcome.raw.safety === 1) return "safety";
10605
+ if (record.splitTag === "holdout") return "holdout";
10606
+ if (record.splitTag === "dev") return "dev";
10607
+ return "practice";
10608
+ }
10609
+ function scoreOf(record) {
10610
+ const score = record.outcome.holdoutScore ?? record.outcome.searchScore;
10611
+ if (typeof score === "number" && Number.isFinite(score)) return clamp014(score);
10612
+ const rawScore = record.outcome.raw.score ?? record.outcome.raw.composite;
10613
+ return typeof rawScore === "number" && Number.isFinite(rawScore) ? clamp014(rawScore) : 0;
10614
+ }
10615
+ function rawPassOf(record) {
10616
+ const rawPass = record.outcome.raw.pass;
10617
+ if (typeof rawPass === "boolean") return rawPass;
10618
+ if (typeof rawPass === "number" && Number.isFinite(rawPass)) return rawPass >= 1;
10619
+ return null;
10620
+ }
10621
+ function passOf(record, score, threshold) {
10622
+ const rawPass = rawPassOf(record);
10623
+ if (rawPass !== null) return rawPass && !record.failureMode;
10624
+ return score >= threshold && !record.failureMode;
10625
+ }
10626
+ function failureModeOf(record, score, threshold) {
10627
+ if (record.failureMode) return record.failureMode;
10628
+ const belowThreshold = `quality-below-threshold: ${Math.round(score * 100)}% < ${Math.round(threshold * 100)}%`;
10629
+ if (rawPassOf(record) === false) {
10630
+ return score < threshold ? belowThreshold : "product-pass-failed";
10631
+ }
10632
+ if (passOf(record, score, threshold)) return null;
10633
+ return belowThreshold;
10634
+ }
10635
+ function numericDimensions(record) {
10636
+ const dimensions = {};
10637
+ for (const [key, value] of Object.entries(record.outcome.raw ?? {})) {
10638
+ if (typeof value === "number" && Number.isFinite(value)) dimensions[key] = value;
10639
+ }
10640
+ if (Object.keys(dimensions).length === 0) dimensions.score = scoreOf(record);
10641
+ return dimensions;
10642
+ }
10643
+ function armIdOf(record) {
10644
+ if (record.candidateId) return record.candidateId;
10645
+ const variant = record.agentProfile?.dimensions?.variant;
10646
+ if (typeof variant === "string" && variant.length > 0) return variant;
10647
+ const variantId = record.agentProfile?.dimensions?.variantId;
10648
+ if (typeof variantId === "string" && variantId.length > 0) return variantId;
10649
+ return "production-profile";
10650
+ }
10651
+ function backendOf(record) {
10652
+ const backend = record.agentProfile?.dimensions?.backend;
10653
+ return typeof backend === "string" && backend.length > 0 ? backend : "unknown";
10654
+ }
10655
+ function modelProvider(record) {
10656
+ const backend = backendOf(record);
10657
+ if (backend === "cli-bridge") return "cli-bridge";
10658
+ if (backend === "sandbox") return "router";
10659
+ if (record.model.startsWith("router/")) return "router";
10660
+ return backend === "unknown" ? "router" : backend;
10661
+ }
10662
+ function runtimeResolution(record, opts) {
10663
+ const reasoningEffort = record.agentProfile?.dimensions?.reasoningLevel;
10664
+ return {
10665
+ model: record.agentProfile?.model ?? record.model,
10666
+ harness: record.agentProfile?.harness?.id ?? `${opts.projectId}-canonical-eval`,
10667
+ backend: backendOf(record),
10668
+ ...typeof reasoningEffort === "string" && reasoningEffort.length > 0 ? { reasoningEffort } : {}
10669
+ };
10670
+ }
10671
+ function sourceProfileHash(record) {
10672
+ return record.agentProfile?.sourceProfile?.hash ?? record.agentProfile?.cellId ?? sha256(
10673
+ JSON.stringify({
10674
+ candidateId: record.candidateId,
10675
+ promptHash: record.promptHash,
10676
+ configHash: record.configHash
10677
+ })
10678
+ );
10679
+ }
10680
+ function materializedProfileHash(record, runtime) {
10681
+ return sha256(
10682
+ JSON.stringify({ sourceProfileHash: sourceProfileHash(record), model: runtime.model })
10683
+ );
10684
+ }
10685
+ function profileIdOf(record, armId, runtime, opts) {
10686
+ const base = record.agentProfile?.profileId ?? opts.fallbackProfileId ?? armId;
10687
+ const withArm = base.endsWith(`:${armId}`) ? base : `${base}:${armId}`;
10688
+ return `${withArm}:${runtime.model}`;
10689
+ }
10690
+ function toolCallsOf(record, runDir, opts) {
10691
+ const raw = record.outcome.raw;
10692
+ const primary = Number(raw.tool_call_count ?? raw.tool_calls ?? raw.toolCalls ?? 0);
10693
+ if (Number.isFinite(primary) && primary > 0) return primary;
10694
+ const fallback = opts.toolCallFallback?.(record, runDir);
10695
+ if (fallback !== void 0 && Number.isFinite(fallback) && fallback > 0) return fallback;
10696
+ return Number.isFinite(primary) ? primary : 0;
10697
+ }
10698
+ var TRACE_CANDIDATES = ["traces.jsonl", "traces", "trace", "trace-store"];
10699
+ var RAW_CANDIDATES = ["raws.jsonl", "raw-events"];
10700
+ var SCORE_CANDIDATES = ["scores.json", "manifest.json"];
10701
+ function firstExisting(runDir, candidates) {
10702
+ return candidates.find((candidate) => existsSync5(join4(runDir, candidate))) ?? candidates[0];
10703
+ }
10704
+ function relativeArtifacts(runDir) {
10705
+ return {
10706
+ records: "records.jsonl",
10707
+ traces: firstExisting(runDir, TRACE_CANDIDATES),
10708
+ raws: firstExisting(runDir, RAW_CANDIDATES),
10709
+ scores: firstExisting(runDir, SCORE_CANDIDATES),
10710
+ workspace: existsSync5(join4(runDir, "workspace")) ? "workspace" : "."
10711
+ };
10712
+ }
10713
+ function absoluteArtifacts(runDir) {
10714
+ const rel = relativeArtifacts(runDir);
10715
+ return {
10716
+ records: resolve(runDir, rel.records),
10717
+ traces: resolve(runDir, rel.traces),
10718
+ raws: resolve(runDir, rel.raws),
10719
+ scores: resolve(runDir, rel.scores),
10720
+ workspace: resolve(runDir, rel.workspace)
10721
+ };
10722
+ }
10723
+ function materializeRunArtifacts(runDir, outDir, index) {
10724
+ if (resolve(runDir) === resolve(outDir)) return relativeArtifacts(runDir);
10725
+ const rel = relative(runDir, outDir);
10726
+ if (rel && !rel.startsWith("..") && !isAbsolute(rel)) {
10727
+ throw new ValidationError(`outDir must not be nested inside runDir: ${outDir}`);
10728
+ }
10729
+ const id = `${String(index + 1).padStart(2, "0")}-${safePathPart(basename2(runDir))}-${sha256(runDir).slice(7, 19)}`;
10730
+ const destRel = join4("source-runs", id);
10731
+ const destAbs = join4(outDir, destRel);
10732
+ mkdirSync3(dirname3(destAbs), { recursive: true });
10733
+ cpSync(runDir, destAbs, { recursive: true, force: true });
10734
+ const source = relativeArtifacts(runDir);
10735
+ return {
10736
+ records: join4(destRel, source.records),
10737
+ traces: join4(destRel, source.traces),
10738
+ raws: join4(destRel, source.raws),
10739
+ scores: join4(destRel, source.scores),
10740
+ workspace: source.workspace === "." ? destRel : join4(destRel, source.workspace)
10741
+ };
10742
+ }
10743
+ function existsArtifact(artifactRoot, value) {
10744
+ return existsSync5(resolve(artifactRoot, value));
10745
+ }
10746
+ function runRecordToProductBenchmarkRecord(record, runDir, artifactRoot, artifacts, options) {
10747
+ const opts = resolveOptions(options);
10748
+ const runtime = runtimeResolution(record, opts);
10749
+ const score = scoreOf(record);
10750
+ const armId = armIdOf(record);
10751
+ const inputTokens = record.tokenUsage.input;
10752
+ const outputTokens = record.tokenUsage.output;
10753
+ const toolCallCount = toolCallsOf(record, runDir, opts);
10754
+ const dimensions = numericDimensions(record);
10755
+ if (!("tool_calls" in dimensions)) dimensions.tool_calls = toolCallCount;
10756
+ const product = {
10757
+ schemaVersion: 1,
10758
+ projectId: opts.projectId,
10759
+ benchmarkId: opts.benchmarkId,
10760
+ runId: record.runId,
10761
+ scenarioId: record.scenarioId ?? record.experimentId,
10762
+ split: splitOf(record, opts),
10763
+ armId,
10764
+ rep: Number(record.seed ?? 0) + 1,
10765
+ agentProfile: {
10766
+ id: profileIdOf(record, armId, runtime, opts),
10767
+ hash: materializedProfileHash(record, runtime),
10768
+ path: opts.agentProfilePath,
10769
+ declared: runtime,
10770
+ resolved: runtime
10771
+ },
10772
+ model: { provider: modelProvider(record), id: record.model },
10773
+ backend: { kind: backendOf(record), version: opts.backendVersion },
10774
+ outcome: {
10775
+ pass: passOf(record, score, opts.passThreshold),
10776
+ score,
10777
+ dimensions,
10778
+ failureMode: failureModeOf(record, score, opts.passThreshold)
10779
+ },
10780
+ usage: {
10781
+ inputTokens,
10782
+ outputTokens,
10783
+ costUsd: record.costUsd,
10784
+ // Rounded: the bundle contract requires integer milliseconds.
10785
+ wallMs: Math.round(record.wallMs),
10786
+ toolCalls: toolCallCount
10787
+ },
10788
+ integrity: {
10789
+ realBackend: inputTokens + outputTokens > 0,
10790
+ rawCapture: existsArtifact(artifactRoot, artifacts.raws),
10791
+ traceCapture: existsArtifact(artifactRoot, artifacts.traces),
10792
+ noStubRows: inputTokens + outputTokens > 0,
10793
+ priced: record.costUsd > 0,
10794
+ profileMaterialized: Boolean(record.agentProfile?.cellId)
10795
+ },
10796
+ artifacts
10797
+ };
10798
+ return validateProductBenchmarkRecord(product);
10799
+ }
10800
+ function uniqueArmRecords(records) {
10801
+ const byArm = /* @__PURE__ */ new Map();
10802
+ for (const record of records) {
10803
+ const prior = byArm.get(record.armId);
10804
+ if (!prior) {
10805
+ byArm.set(record.armId, record);
10806
+ continue;
10807
+ }
10808
+ const fields = [
10809
+ ["profileId", prior.agentProfile.id, record.agentProfile.id],
10810
+ ["model", prior.agentProfile.resolved.model, record.agentProfile.resolved.model],
10811
+ ["backend", prior.agentProfile.resolved.backend, record.agentProfile.resolved.backend],
10812
+ ["harness", prior.agentProfile.resolved.harness, record.agentProfile.resolved.harness],
10813
+ [
10814
+ "reasoningEffort",
10815
+ prior.agentProfile.resolved.reasoningEffort,
10816
+ record.agentProfile.resolved.reasoningEffort
10817
+ ]
10818
+ ];
10819
+ for (const [name, a, b] of fields) {
10820
+ if (a !== b) {
10821
+ throw new ValidationError(
10822
+ `records for arm '${record.armId}' disagree on ${name} (${String(a)} vs ${String(b)}) \u2014 one arm id must map to one policy`
10823
+ );
10824
+ }
10825
+ }
10826
+ }
10827
+ return [...byArm.values()];
10828
+ }
10829
+ function uniqueScenarioRecords(records) {
10830
+ const byScenario = /* @__PURE__ */ new Map();
10831
+ for (const record of records) {
10832
+ const prior = byScenario.get(record.scenarioId);
10833
+ if (!prior) {
10834
+ byScenario.set(record.scenarioId, record);
10835
+ continue;
10836
+ }
10837
+ if (prior.split !== record.split) {
10838
+ throw new ValidationError(
10839
+ `records for scenario '${record.scenarioId}' disagree on split (${prior.split} vs ${record.split})`
10840
+ );
10841
+ }
10842
+ }
10843
+ return [...byScenario.values()];
10844
+ }
10845
+ function buildProductBenchmarkManifest(records, options) {
10846
+ if (records.length === 0) {
10847
+ throw new ValidationError("cannot build a product benchmark manifest from zero records");
10848
+ }
10849
+ const scenarioTagPrefix = options.scenarioTagPrefix ?? defaultScenarioTagPrefix(options.projectId);
10850
+ const mutableSurfaces = options.mutableSurfaces ?? productBenchmarkMutableSurfaces;
10851
+ const byProfile = /* @__PURE__ */ new Map();
10852
+ for (const record of records) {
10853
+ byProfile.set(record.agentProfile.id, {
10854
+ id: record.agentProfile.id,
10855
+ profileHash: record.agentProfile.hash,
10856
+ agentProfilePath: record.agentProfile.path
10857
+ });
10858
+ }
10859
+ const manifest = {
10860
+ schemaVersion: 1,
10861
+ projectId: options.projectId,
10862
+ benchmarkId: options.benchmarkId,
10863
+ repo: productBenchmarkRepoIdentity(),
10864
+ substrate: {
10865
+ agentEval: packageVersion("@tangle-network/agent-eval"),
10866
+ agentRuntime: packageVersion("@tangle-network/agent-runtime"),
10867
+ agentInterface: packageVersion("@tangle-network/agent-interface"),
10868
+ sandbox: packageVersion("@tangle-network/sandbox"),
10869
+ ...options.substrate
10870
+ },
10871
+ profiles: [...byProfile.values()],
10872
+ arms: uniqueArmRecords(records).map((record) => ({
10873
+ id: record.armId,
10874
+ profileId: record.agentProfile.id,
10875
+ mutableSurfaces: [...mutableSurfaces],
10876
+ policyAxes: {
10877
+ carrier: record.armId.includes("policy") ? "resource-file" : "profile",
10878
+ model: record.agentProfile.resolved.model,
10879
+ backend: record.agentProfile.resolved.backend,
10880
+ harness: record.agentProfile.resolved.harness,
10881
+ ...record.agentProfile.resolved.reasoningEffort !== void 0 ? { reasoningEffort: record.agentProfile.resolved.reasoningEffort } : {}
10882
+ }
10883
+ })),
10884
+ scenarios: uniqueScenarioRecords(records).map((record) => ({
10885
+ id: record.scenarioId,
10886
+ split: record.split,
10887
+ tags: [scenarioTagPrefix, options.benchmarkId, record.split],
10888
+ sourceAllowedForSynthesis: false
10889
+ })),
10890
+ budgets: {
10891
+ maxUsd: records.reduce((sum4, record) => sum4 + record.usage.costUsd, 0),
10892
+ maxCells: records.length,
10893
+ maxWallMs: records.reduce((sum4, record) => sum4 + record.usage.wallMs, 0)
10894
+ },
10895
+ expectedArtifactDir: resolve(options.outDir)
10896
+ };
10897
+ return validateProductBenchmarkManifest(manifest);
10898
+ }
10899
+ function exportProductBenchmark(options) {
10900
+ const { runDir, ...rest } = options;
10901
+ return exportProductBenchmarkRuns({ ...rest, runDirs: [runDir] });
10902
+ }
10903
+ function exportProductBenchmarkRuns(options) {
10904
+ const opts = resolveOptions(options);
10905
+ const outDir = resolve(options.outDir);
10906
+ const runDirs = options.runDirs.map((runDir) => resolve(runDir));
10907
+ if (runDirs.length === 0) throw new ValidationError("export requires at least one run directory");
10908
+ const materialize = options.materializeSourceRuns !== false;
10909
+ const rows = runDirs.flatMap((runDir) => {
10910
+ const sourceRecordsPath = join4(runDir, "records.jsonl");
10911
+ if (!existsSync5(sourceRecordsPath)) throw new ValidationError(`missing ${sourceRecordsPath}`);
10912
+ const records = readRunRecords(sourceRecordsPath);
10913
+ if (records.length === 0) throw new ValidationError(`${sourceRecordsPath} is empty`);
10914
+ return records.map((record) => ({ record, runDir }));
10915
+ });
10916
+ mkdirSync3(outDir, { recursive: true });
10917
+ const artifactsByRunDir = /* @__PURE__ */ new Map();
10918
+ for (const [index, runDir] of runDirs.entries()) {
10919
+ artifactsByRunDir.set(
10920
+ runDir,
10921
+ materialize ? materializeRunArtifacts(runDir, outDir, index) : absoluteArtifacts(runDir)
10922
+ );
10923
+ }
10924
+ const artifactRoot = materialize ? outDir : "/";
10925
+ const normalized = rows.map(
10926
+ ({ record, runDir }) => runRecordToProductBenchmarkRecord(
10927
+ record,
10928
+ runDir,
10929
+ artifactRoot,
10930
+ artifactsByRunDir.get(runDir),
10931
+ { ...options, runDirs, backendVersion: opts.backendVersion }
10932
+ )
10933
+ );
10934
+ const manifest = buildProductBenchmarkManifest(normalized, {
10935
+ outDir,
10936
+ projectId: opts.projectId,
10937
+ benchmarkId: opts.benchmarkId,
10938
+ scenarioTagPrefix: opts.scenarioTagPrefix,
10939
+ mutableSurfaces: opts.mutableSurfaces,
10940
+ ...options.substrate !== void 0 ? { substrate: options.substrate } : {}
10941
+ });
10942
+ const manifestPath = join4(outDir, "product-benchmark-manifest.json");
10943
+ const recordsPath = join4(outDir, "product-benchmark-records.jsonl");
10944
+ writeFileSync(manifestPath, `${JSON.stringify(manifest, null, 2)}
10945
+ `);
10946
+ writeFileSync(recordsPath, `${normalized.map((record) => JSON.stringify(record)).join("\n")}
10947
+ `);
10948
+ return { manifestPath, recordsPath, records: normalized.length };
10949
+ }
10950
+ function defaultScenarioTagPrefix(projectId) {
10951
+ return projectId.replace(/-agent$/, "") || projectId;
10952
+ }
10953
+ function resolveOptions(options) {
10954
+ return {
10955
+ projectId: options.projectId,
10956
+ benchmarkId: options.benchmarkId,
10957
+ agentProfilePath: options.agentProfilePath,
10958
+ passThreshold: options.passThreshold ?? 0.7,
10959
+ scenarioTagPrefix: options.scenarioTagPrefix ?? defaultScenarioTagPrefix(options.projectId),
10960
+ ...options.fallbackProfileId !== void 0 ? { fallbackProfileId: options.fallbackProfileId } : {},
10961
+ mutableSurfaces: options.mutableSurfaces ?? productBenchmarkMutableSurfaces,
10962
+ ...options.classifySplit !== void 0 ? { classifySplit: options.classifySplit } : {},
10963
+ ...options.toolCallFallback !== void 0 ? { toolCallFallback: options.toolCallFallback } : {},
10964
+ backendVersion: options.backendVersion ?? packageVersion("@tangle-network/sandbox")
10965
+ };
10966
+ }
10967
+
10968
+ // src/product-benchmark/index.ts
10969
+ var productBenchmarkSplits = ["practice", "dev", "holdout", "safety", "sentinel"];
10970
+ function isObject2(value) {
10971
+ return Boolean(value) && typeof value === "object" && !Array.isArray(value);
10972
+ }
10973
+ function fail(path, message) {
10974
+ throw new ValidationError(`${path}: ${message}`);
10975
+ }
10976
+ function wrapValidationError(path, err) {
10977
+ if (err instanceof ValidationError) {
10978
+ throw new ValidationError(`${path}: ${err.message}`, { cause: err });
10979
+ }
10980
+ throw new ValidationError(`${path}: ${err instanceof Error ? err.message : String(err)}`, {
10981
+ cause: err
10982
+ });
10983
+ }
10984
+ function expectObject(value, path) {
10985
+ if (!isObject2(value)) fail(path, "must be an object");
10986
+ return value;
10987
+ }
10988
+ function expectString(value, path) {
10989
+ if (typeof value !== "string" || value.trim().length === 0)
10990
+ fail(path, "must be a non-empty string");
10991
+ return value;
10992
+ }
10993
+ function expectBoolean(value, path) {
10994
+ if (typeof value !== "boolean") fail(path, "must be a boolean");
10995
+ return value;
10996
+ }
10997
+ function expectNumber(value, path, opts = {}) {
10998
+ if (typeof value !== "number" || !Number.isFinite(value)) fail(path, "must be a finite number");
10999
+ if (opts.integer && !Number.isInteger(value)) fail(path, "must be an integer");
11000
+ if (opts.min !== void 0 && value < opts.min) fail(path, `must be >= ${opts.min}`);
11001
+ if (opts.max !== void 0 && value > opts.max) fail(path, `must be <= ${opts.max}`);
11002
+ return value;
11003
+ }
11004
+ function expectStringArray(value, path) {
11005
+ if (!Array.isArray(value)) fail(path, "must be an array");
11006
+ return value.map((entry, index) => expectString(entry, `${path}[${index}]`));
11007
+ }
11008
+ function expectObjectArray(value, path) {
11009
+ if (!Array.isArray(value)) fail(path, "must be an array");
11010
+ return value.map((entry, index) => expectObject(entry, `${path}[${index}]`));
11011
+ }
11012
+ function expectSplit(value, path) {
11013
+ const split = expectString(value, path);
11014
+ if (!productBenchmarkSplits.includes(split)) {
11015
+ fail(path, `must be one of ${productBenchmarkSplits.join(", ")}`);
11016
+ }
11017
+ return split;
11018
+ }
11019
+ function optionalString(value, path) {
11020
+ if (value === void 0) return void 0;
11021
+ return expectString(value, path);
11022
+ }
11023
+ function expectDimensions(value, path) {
11024
+ const obj = expectObject(value, path);
11025
+ const out = {};
11026
+ for (const [key, raw] of Object.entries(obj)) out[key] = expectNumber(raw, `${path}.${key}`);
11027
+ return out;
11028
+ }
11029
+ function expectRuntimeResolution(value, path) {
11030
+ const obj = expectObject(value, path);
11031
+ return {
11032
+ model: expectString(obj.model, `${path}.model`),
11033
+ harness: expectString(obj.harness, `${path}.harness`),
11034
+ backend: expectString(obj.backend, `${path}.backend`),
11035
+ ...obj.reasoningEffort !== void 0 ? { reasoningEffort: optionalString(obj.reasoningEffort, `${path}.reasoningEffort`) } : {}
11036
+ };
11037
+ }
11038
+ function validateProductBenchmarkManifest(value) {
11039
+ const obj = expectObject(value, "manifest");
11040
+ if (obj.schemaVersion !== 1) fail("manifest.schemaVersion", "must be 1");
11041
+ const repo = expectObject(obj.repo, "manifest.repo");
11042
+ const substrate = expectObject(obj.substrate, "manifest.substrate");
11043
+ const profiles = expectObjectArray(obj.profiles, "manifest.profiles").map((profile, index) => ({
11044
+ id: expectString(profile.id, `manifest.profiles[${index}].id`),
11045
+ profileHash: expectString(profile.profileHash, `manifest.profiles[${index}].profileHash`),
11046
+ agentProfilePath: expectString(
11047
+ profile.agentProfilePath,
11048
+ `manifest.profiles[${index}].agentProfilePath`
11049
+ )
11050
+ }));
11051
+ const arms = expectObjectArray(obj.arms, "manifest.arms").map((arm, index) => ({
11052
+ id: expectString(arm.id, `manifest.arms[${index}].id`),
11053
+ profileId: expectString(arm.profileId, `manifest.arms[${index}].profileId`),
11054
+ mutableSurfaces: expectStringArray(
11055
+ arm.mutableSurfaces,
11056
+ `manifest.arms[${index}].mutableSurfaces`
11057
+ ),
11058
+ policyAxes: expectObject(arm.policyAxes, `manifest.arms[${index}].policyAxes`)
11059
+ }));
11060
+ const scenarios = expectObjectArray(obj.scenarios, "manifest.scenarios").map(
11061
+ (scenario, index) => ({
11062
+ id: expectString(scenario.id, `manifest.scenarios[${index}].id`),
11063
+ split: expectSplit(scenario.split, `manifest.scenarios[${index}].split`),
11064
+ tags: expectStringArray(scenario.tags, `manifest.scenarios[${index}].tags`),
11065
+ sourceAllowedForSynthesis: expectBoolean(
11066
+ scenario.sourceAllowedForSynthesis,
11067
+ `manifest.scenarios[${index}].sourceAllowedForSynthesis`
11068
+ )
11069
+ })
11070
+ );
11071
+ const budgets = expectObject(obj.budgets, "manifest.budgets");
11072
+ if (profiles.length === 0) fail("manifest.profiles", "must contain at least one profile");
11073
+ if (arms.length === 0) fail("manifest.arms", "must contain at least one arm");
11074
+ if (scenarios.length === 0) fail("manifest.scenarios", "must contain at least one scenario");
11075
+ assertUnique(
11076
+ profiles.map((profile) => profile.id),
11077
+ "manifest.profiles.id"
11078
+ );
11079
+ assertUnique(
11080
+ arms.map((arm) => arm.id),
11081
+ "manifest.arms.id"
11082
+ );
11083
+ assertUnique(
11084
+ scenarios.map((scenario) => scenario.id),
11085
+ "manifest.scenarios.id"
11086
+ );
11087
+ const profileIds = new Set(profiles.map((profile) => profile.id));
11088
+ for (const arm of arms) {
11089
+ if (!profileIds.has(arm.profileId))
11090
+ fail(`manifest.arms.${arm.id}.profileId`, `unknown profile ${arm.profileId}`);
11091
+ }
11092
+ return {
11093
+ schemaVersion: 1,
11094
+ projectId: expectString(obj.projectId, "manifest.projectId"),
11095
+ benchmarkId: expectString(obj.benchmarkId, "manifest.benchmarkId"),
11096
+ repo: {
11097
+ url: expectString(repo.url, "manifest.repo.url"),
11098
+ commit: expectString(repo.commit, "manifest.repo.commit"),
11099
+ branch: expectString(repo.branch, "manifest.repo.branch")
11100
+ },
11101
+ substrate: {
11102
+ agentEval: expectString(substrate.agentEval, "manifest.substrate.agentEval"),
11103
+ agentRuntime: expectString(substrate.agentRuntime, "manifest.substrate.agentRuntime"),
11104
+ agentInterface: expectString(substrate.agentInterface, "manifest.substrate.agentInterface"),
11105
+ sandbox: expectString(substrate.sandbox, "manifest.substrate.sandbox"),
11106
+ ...substrate.agentBench !== void 0 ? { agentBench: expectString(substrate.agentBench, "manifest.substrate.agentBench") } : {}
11107
+ },
11108
+ profiles,
11109
+ arms,
11110
+ scenarios,
11111
+ budgets: {
11112
+ maxUsd: expectNumber(budgets.maxUsd, "manifest.budgets.maxUsd", { min: 0 }),
11113
+ maxCells: expectNumber(budgets.maxCells, "manifest.budgets.maxCells", {
11114
+ min: 0,
11115
+ integer: true
11116
+ }),
11117
+ maxWallMs: expectNumber(budgets.maxWallMs, "manifest.budgets.maxWallMs", {
11118
+ min: 0,
11119
+ integer: true
11120
+ })
11121
+ },
11122
+ expectedArtifactDir: expectString(obj.expectedArtifactDir, "manifest.expectedArtifactDir")
11123
+ };
11124
+ }
11125
+ function validateProductBenchmarkRecord(value) {
11126
+ const obj = expectObject(value, "record");
11127
+ if (obj.schemaVersion !== 1) fail("record.schemaVersion", "must be 1");
11128
+ const agentProfile = expectObject(obj.agentProfile, "record.agentProfile");
11129
+ const model = expectObject(obj.model, "record.model");
11130
+ const backend = expectObject(obj.backend, "record.backend");
11131
+ const outcome = expectObject(obj.outcome, "record.outcome");
11132
+ const usage = expectObject(obj.usage, "record.usage");
11133
+ const integrity = expectObject(obj.integrity, "record.integrity");
11134
+ const artifacts = expectObject(obj.artifacts, "record.artifacts");
11135
+ const record = {
11136
+ schemaVersion: 1,
11137
+ projectId: expectString(obj.projectId, "record.projectId"),
11138
+ benchmarkId: expectString(obj.benchmarkId, "record.benchmarkId"),
11139
+ runId: expectString(obj.runId, "record.runId"),
11140
+ scenarioId: expectString(obj.scenarioId, "record.scenarioId"),
11141
+ split: expectSplit(obj.split, "record.split"),
11142
+ armId: expectString(obj.armId, "record.armId"),
11143
+ rep: expectNumber(obj.rep, "record.rep", { min: 1, integer: true }),
11144
+ agentProfile: {
11145
+ id: expectString(agentProfile.id, "record.agentProfile.id"),
11146
+ hash: expectString(agentProfile.hash, "record.agentProfile.hash"),
11147
+ path: expectString(agentProfile.path, "record.agentProfile.path"),
11148
+ declared: expectRuntimeResolution(agentProfile.declared, "record.agentProfile.declared"),
11149
+ resolved: expectRuntimeResolution(agentProfile.resolved, "record.agentProfile.resolved")
11150
+ },
11151
+ model: {
11152
+ provider: expectString(model.provider, "record.model.provider"),
11153
+ id: expectString(model.id, "record.model.id")
11154
+ },
11155
+ backend: {
11156
+ kind: expectString(backend.kind, "record.backend.kind"),
11157
+ version: expectString(backend.version, "record.backend.version")
11158
+ },
11159
+ outcome: {
11160
+ pass: expectBoolean(outcome.pass, "record.outcome.pass"),
11161
+ score: expectNumber(outcome.score, "record.outcome.score", { min: 0, max: 1 }),
11162
+ dimensions: expectDimensions(outcome.dimensions, "record.outcome.dimensions"),
11163
+ failureMode: outcome.failureMode === null ? null : expectString(outcome.failureMode, "record.outcome.failureMode")
11164
+ },
11165
+ usage: {
11166
+ inputTokens: expectNumber(usage.inputTokens, "record.usage.inputTokens", {
11167
+ min: 0,
11168
+ integer: true
11169
+ }),
11170
+ outputTokens: expectNumber(usage.outputTokens, "record.usage.outputTokens", {
11171
+ min: 0,
11172
+ integer: true
11173
+ }),
11174
+ costUsd: expectNumber(usage.costUsd, "record.usage.costUsd", { min: 0 }),
11175
+ wallMs: expectNumber(usage.wallMs, "record.usage.wallMs", { min: 0, integer: true }),
11176
+ toolCalls: expectNumber(usage.toolCalls, "record.usage.toolCalls", { min: 0, integer: true })
11177
+ },
11178
+ integrity: {
11179
+ realBackend: expectBoolean(integrity.realBackend, "record.integrity.realBackend"),
11180
+ rawCapture: expectBoolean(integrity.rawCapture, "record.integrity.rawCapture"),
11181
+ traceCapture: expectBoolean(integrity.traceCapture, "record.integrity.traceCapture"),
11182
+ noStubRows: expectBoolean(integrity.noStubRows, "record.integrity.noStubRows"),
11183
+ priced: expectBoolean(integrity.priced, "record.integrity.priced"),
11184
+ profileMaterialized: expectBoolean(
11185
+ integrity.profileMaterialized,
11186
+ "record.integrity.profileMaterialized"
11187
+ )
11188
+ },
11189
+ artifacts: {
11190
+ records: expectString(artifacts.records, "record.artifacts.records"),
11191
+ traces: expectString(artifacts.traces, "record.artifacts.traces"),
11192
+ raws: expectString(artifacts.raws, "record.artifacts.raws"),
11193
+ scores: expectString(artifacts.scores, "record.artifacts.scores"),
11194
+ workspace: expectString(artifacts.workspace, "record.artifacts.workspace")
11195
+ }
11196
+ };
11197
+ if (record.integrity.realBackend && record.usage.inputTokens + record.usage.outputTokens === 0) {
11198
+ fail("record.usage", "realBackend rows must carry non-zero token usage");
11199
+ }
11200
+ return record;
11201
+ }
11202
+ function productBenchmarkIntegrityFailures(record) {
11203
+ const failures = [];
11204
+ for (const [key, value] of Object.entries(record.integrity)) {
11205
+ if (!value) failures.push(`${record.runId}:${key}=false`);
11206
+ }
11207
+ return failures;
11208
+ }
11209
+ function readProductBenchmarkRecords(path) {
11210
+ const lines = readFileSync7(path, "utf8").split("\n").map((line) => line.trim()).filter(Boolean);
11211
+ const records = [];
11212
+ for (const [index, line] of lines.entries()) {
11213
+ try {
11214
+ records.push(validateProductBenchmarkRecord(JSON.parse(line)));
11215
+ } catch (err) {
11216
+ wrapValidationError(`${path}:${index + 1}`, err);
11217
+ }
11218
+ }
11219
+ return records;
11220
+ }
11221
+ function readProductBenchmarkManifest(path) {
11222
+ try {
11223
+ return validateProductBenchmarkManifest(JSON.parse(readFileSync7(path, "utf8")));
11224
+ } catch (err) {
11225
+ wrapValidationError(path, err);
11226
+ }
11227
+ }
11228
+ function validateProductBenchmarkRun(input) {
11229
+ const manifest = readProductBenchmarkManifest(input.manifestPath);
11230
+ const records = readProductBenchmarkRecords(input.recordsPath);
11231
+ const manifestProjectBench = `${manifest.projectId}/${manifest.benchmarkId}`;
11232
+ const integrityFailures = records.flatMap(productBenchmarkIntegrityFailures);
11233
+ const missingArtifacts = input.checkArtifacts === false ? [] : records.flatMap(
11234
+ (record) => missingArtifactsForRecord(record, input.artifactRoot ?? dirname4(input.recordsPath))
11235
+ );
11236
+ for (const [index, record] of records.entries()) {
11237
+ const recordProjectBench = `${record.projectId}/${record.benchmarkId}`;
11238
+ if (recordProjectBench !== manifestProjectBench) {
11239
+ fail(
11240
+ `records[${index}]`,
11241
+ `project/benchmark ${recordProjectBench} does not match manifest ${manifestProjectBench}`
11242
+ );
11243
+ }
11244
+ if (!manifest.arms.some((arm) => arm.id === record.armId))
11245
+ fail(`records[${index}].armId`, `unknown arm ${record.armId}`);
11246
+ if (!manifest.scenarios.some((scenario) => scenario.id === record.scenarioId)) {
11247
+ fail(`records[${index}].scenarioId`, `unknown scenario ${record.scenarioId}`);
11248
+ }
11249
+ }
11250
+ const repoFailures = ["url", "commit", "branch"].filter((key) => manifest.repo[key].trim().length === 0 || manifest.repo[key] === "unknown").map((key) => `manifest.repo.${key}`);
11251
+ const substrateFailures = ["agentEval", "agentRuntime", "agentInterface", "sandbox"].filter(
11252
+ (key) => manifest.substrate[key].trim().length === 0 || manifest.substrate[key] === "unknown"
11253
+ ).map((key) => `manifest.substrate.${key}`);
11254
+ return {
11255
+ manifestPath: input.manifestPath,
11256
+ recordsPath: input.recordsPath,
11257
+ records: records.length,
11258
+ repoFailures,
11259
+ substrateFailures,
11260
+ projects: sortedUnique(records.map((record) => record.projectId)),
11261
+ benchmarks: sortedUnique(records.map((record) => record.benchmarkId)),
11262
+ arms: sortedUnique(records.map((record) => record.armId)),
11263
+ scenarios: sortedUnique(records.map((record) => record.scenarioId)),
11264
+ passed: records.filter((record) => record.outcome.pass).length,
11265
+ failed: records.filter((record) => !record.outcome.pass).length,
11266
+ inputTokens: sum3(records, (record) => record.usage.inputTokens),
11267
+ outputTokens: sum3(records, (record) => record.usage.outputTokens),
11268
+ costUsd: sum3(records, (record) => record.usage.costUsd),
11269
+ wallMs: sum3(records, (record) => record.usage.wallMs),
11270
+ integrityFailures,
11271
+ missingArtifacts
11272
+ };
11273
+ }
11274
+ function missingArtifactsForRecord(record, artifactRoot) {
11275
+ const missing = [];
11276
+ for (const [key, value] of Object.entries(record.artifacts)) {
11277
+ const path = resolveArtifactPath(value, artifactRoot);
11278
+ if (!existsSync6(path)) missing.push(`${record.runId}:${key}:${value}`);
11279
+ }
11280
+ return missing;
11281
+ }
11282
+ function resolveArtifactPath(value, artifactRoot) {
11283
+ return value.startsWith("/") ? value : join5(artifactRoot, value);
11284
+ }
11285
+ function assertUnique(values, path) {
11286
+ const seen = /* @__PURE__ */ new Set();
11287
+ for (const value of values) {
11288
+ if (seen.has(value)) fail(path, `duplicate ${value}`);
11289
+ seen.add(value);
11290
+ }
11291
+ }
11292
+ function sortedUnique(values) {
11293
+ return [...new Set(values)].sort();
11294
+ }
11295
+ function sum3(items, fn) {
11296
+ return items.reduce((total, item) => total + fn(item), 0);
11297
+ }
11298
+ function findProductBenchmarkArtifacts(runDir) {
11299
+ const manifestPath = join5(runDir, "product-benchmark-manifest.json");
11300
+ const recordsPath = join5(runDir, "product-benchmark-records.jsonl");
11301
+ if (existsSync6(manifestPath) && statSync3(manifestPath).isFile() && existsSync6(recordsPath) && statSync3(recordsPath).isFile()) {
11302
+ return { manifestPath, recordsPath };
11303
+ }
11304
+ return null;
11305
+ }
11306
+ function assertProductBenchmarkRun(runDir) {
11307
+ const artifacts = findProductBenchmarkArtifacts(runDir);
11308
+ if (!artifacts) {
11309
+ fail(runDir, "missing product-benchmark-manifest.json or product-benchmark-records.jsonl");
11310
+ }
11311
+ const report = validateProductBenchmarkRun({ ...artifacts, artifactRoot: runDir });
11312
+ const failures = [
11313
+ ...report.repoFailures,
11314
+ ...report.substrateFailures,
11315
+ ...report.integrityFailures,
11316
+ ...report.missingArtifacts
11317
+ ];
11318
+ if (failures.length > 0) {
11319
+ fail(runDir, `product benchmark validation failed:
11320
+ ${failures.join("\n")}`);
11321
+ }
11322
+ return report;
11323
+ }
10220
11324
  export {
10221
11325
  AGENT_PROFILE_KINDS,
10222
11326
  ATTESTATION_ALGORITHM,
@@ -10375,7 +11479,6 @@ export {
10375
11479
  assertNoHiddenLeak,
10376
11480
  assertProductBenchmarkRun,
10377
11481
  assertRealBackend,
10378
- assertRecordIntegrity,
10379
11482
  assertReleaseConfidence,
10380
11483
  assertRunAgentProfileCell,
10381
11484
  assertRunCaptured,
@@ -10421,11 +11524,9 @@ export {
10421
11524
  causalAttribution,
10422
11525
  checkBehavioralCanary,
10423
11526
  checkCanaries,
10424
- checkRecordIntegrity,
10425
11527
  checkSlos,
10426
11528
  checkTraceContracts,
10427
11529
  clamp01,
10428
- classifyEuAiRisk,
10429
11530
  classifyFailure,
10430
11531
  classifyTreatment,
10431
11532
  cliffsDelta,
@@ -10504,7 +11605,6 @@ export {
10504
11605
  errorStreakDetector,
10505
11606
  estimateCost,
10506
11607
  estimateTokens,
10507
- euAiActReport,
10508
11608
  evaluateActionPolicy,
10509
11609
  evaluateContract,
10510
11610
  evaluateHypothesis,
@@ -10513,7 +11613,6 @@ export {
10513
11613
  evaluateReleaseConfidence,
10514
11614
  evaluateTraceContract,
10515
11615
  executeScenario,
10516
- expandMatrix,
10517
11616
  expandProfileAxes,
10518
11617
  expectAgent,
10519
11618
  exportProductBenchmark,
@@ -10552,7 +11651,6 @@ export {
10552
11651
  formatFindings,
10553
11652
  formatScorecardDiff,
10554
11653
  gainHistogram,
10555
- gatePerf,
10556
11654
  gateTreatmentApplied,
10557
11655
  gateTreatmentFromMetrics,
10558
11656
  gateTreatmentFromSpans,
@@ -10632,7 +11730,6 @@ export {
10632
11730
  modelPriceKey,
10633
11731
  mulberry32,
10634
11732
  multiToolchainLayer,
10635
- nistAiRmfReport,
10636
11733
  noProgressDetector,
10637
11734
  normalizeScores,
10638
11735
  notBlocked,
@@ -10697,7 +11794,6 @@ export {
10697
11794
  referenceReplayScenarioToRunScore,
10698
11795
  regexMatch,
10699
11796
  regexMatches,
10700
- renderMarkdown,
10701
11797
  renderMarkdownReport,
10702
11798
  renderPlaybookMarkdown,
10703
11799
  renderPreferenceMemoryMarkdown,
@@ -10745,7 +11841,6 @@ export {
10745
11841
  runsForScenario,
10746
11842
  scalarScore,
10747
11843
  scanForMuffledGates,
10748
- scenarioKey,
10749
11844
  scoreContinuity,
10750
11845
  scoreFromEvals,
10751
11846
  scoreKnowledgeReadiness,
@@ -10762,7 +11857,6 @@ export {
10762
11857
  sentenceReorderMutator,
10763
11858
  serializeFeedbackTrajectoriesJsonl,
10764
11859
  signManifest,
10765
- soc2Report,
10766
11860
  spearmanR,
10767
11861
  splitGold,
10768
11862
  statusAdvanced,
@@ -10771,12 +11865,10 @@ export {
10771
11865
  stringField,
10772
11866
  stripFencedJson,
10773
11867
  subjectiveEval,
10774
- summarize,
10775
11868
  summarizeBackendIntegrity,
10776
11869
  summarizeHarnessResults,
10777
11870
  summarizePrReviewBenchmark,
10778
11871
  summarizePreferenceMemory,
10779
- summarizeRecords,
10780
11872
  summaryTable,
10781
11873
  testJudge,
10782
11874
  textInSnapshot,