@tangle-network/agent-eval 0.120.1 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +6 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4972
  125. package/dist/contract/index.js +0 -1654
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,2152 +0,0 @@
1
- import {
2
- fromClaudeCodeSession,
3
- fromCodexSession,
4
- fromKimiCodeSession,
5
- fromOpenCodeSession,
6
- fromPiSession
7
- } from "../chunk-HKUCJ437.js";
8
- import {
9
- calibrationFromPairs
10
- } from "../chunk-NPCTHQIO.js";
11
- import {
12
- projectRuntimeTrajectoryEvidence
13
- } from "../chunk-T4SQEITX.js";
14
- import {
15
- offPolicyEstimateAll
16
- } from "../chunk-DTJ6QUQB.js";
17
- import {
18
- confidenceInterval
19
- } from "../chunk-PJQFMIOX.js";
20
- import "../chunk-VI2UW6B6.js";
21
- import "../chunk-PXE2VKMX.js";
22
- import {
23
- ValidationError
24
- } from "../chunk-ONWEPEDO.js";
25
- import "../chunk-PZ5AY32C.js";
26
-
27
- // src/belief-state/calibration.ts
28
- function calibrateBeliefDecisions(points, options = {}) {
29
- const filtered = filterCalibrationRegion(points, options);
30
- const pairs = filtered.filter((point) => typeof point.confidence === "number" && point.outcome).map((point) => ({
31
- evalScore: point.confidence,
32
- outcome: outcomeScore(point)
33
- })).filter((pair) => Number.isFinite(pair.outcome));
34
- const minPairs = options.minPairs ?? 10;
35
- if (pairs.length < minPairs) return null;
36
- return calibrationFromPairs(pairs, "belief-confidence", "decision-outcome", {
37
- bins: options.bins ?? 5,
38
- range: { lo: 0, hi: 1 }
39
- });
40
- }
41
- function filterCalibrationRegion(points, options) {
42
- const region = options.region ?? "all";
43
- if (region === "all") return points;
44
- const policy = options.policy;
45
- if (!policy) {
46
- throw new ValidationError(
47
- `calibrateBeliefDecisions: policy is required when region is "${region}"`
48
- );
49
- }
50
- return points.filter((point) => {
51
- const accepted = policy.decide(point).action === "accept";
52
- return region === "accepted" ? accepted : !accepted;
53
- });
54
- }
55
- function outcomeScore(point) {
56
- if (typeof point.outcome?.reward === "number") return point.outcome.reward;
57
- if (typeof point.outcome?.score === "number") return point.outcome.score;
58
- if (point.outcome?.success === true) return 1;
59
- if (point.outcome?.success === false) return 0;
60
- return Number.NaN;
61
- }
62
-
63
- // src/belief-state/ope.ts
64
- function embeddedBeliefOpeTargetPolicy(id = "embedded-target-prob") {
65
- return {
66
- id,
67
- targetProbOf(point) {
68
- return point.targetProb;
69
- },
70
- qHatOf(point) {
71
- return point.qHat;
72
- }
73
- };
74
- }
75
- function beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options = {}) {
76
- const trajectories = [];
77
- const diagnostics = [];
78
- for (const point of points) {
79
- if (!point.outcome) {
80
- diagnostics.push(`${point.id}: missing outcome`);
81
- continue;
82
- }
83
- if (!isBehaviorProbability(point.behaviorProb)) {
84
- diagnostics.push(`${point.id}: invalid behaviorProb ${formatProbability(point.behaviorProb)}`);
85
- continue;
86
- }
87
- let targetProb;
88
- let qHat;
89
- try {
90
- targetProb = targetPolicy.targetProbOf(point);
91
- qHat = targetPolicy.qHatOf?.(point);
92
- } catch (error) {
93
- diagnostics.push(
94
- `${point.id}: target policy ${targetPolicy.id} threw (${errorMessage(error)})`
95
- );
96
- continue;
97
- }
98
- if (!isTargetProbability(targetProb)) {
99
- diagnostics.push(`${point.id}: invalid targetProb ${formatProbability(targetProb)}`);
100
- continue;
101
- }
102
- if (qHat !== null && qHat !== void 0 && !isTargetProbability(qHat)) {
103
- diagnostics.push(`${point.id}: invalid qHat ${formatProbability(qHat)}; ignoring qHat`);
104
- qHat = null;
105
- }
106
- trajectories.push({
107
- runId: point.id,
108
- reward: rewardOf(point),
109
- behaviorProb: point.behaviorProb,
110
- targetProb,
111
- qHat
112
- });
113
- }
114
- return {
115
- targetPolicyId: targetPolicy.id,
116
- trajectories,
117
- dropped: points.length - trajectories.length,
118
- diagnostics: compactDiagnostics(diagnostics, options.maxDiagnostics ?? 20)
119
- };
120
- }
121
- function evaluateBeliefOffPolicy(points, targetPolicy, options = {}) {
122
- const trajectoryReport = beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options);
123
- const { trajectories } = trajectoryReport;
124
- const estimates = offPolicyEstimateAll(trajectories, options);
125
- const support = supportDiagnostics(estimates.dr, {
126
- minEffectiveSampleSize: options.minEffectiveSampleSize ?? 30,
127
- minEffectiveSampleRatio: options.minEffectiveSampleRatio ?? 0.25,
128
- dropped: trajectoryReport.dropped,
129
- diagnostics: trajectoryReport.diagnostics
130
- });
131
- return { targetPolicyId: targetPolicy.id, ...estimates, support };
132
- }
133
- function supportDiagnostics(estimate, options) {
134
- const ratio2 = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
135
- const reasons = [...options.diagnostics];
136
- if (estimate.n === 0) {
137
- reasons.push("no valid OPE trajectories");
138
- }
139
- if (options.dropped > 0) {
140
- reasons.push(`dropped ${options.dropped} unsupported decision(s)`);
141
- }
142
- if (estimate.effectiveSampleSize < options.minEffectiveSampleSize) {
143
- reasons.push(
144
- `effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`
145
- );
146
- }
147
- if (ratio2 < options.minEffectiveSampleRatio) {
148
- reasons.push(
149
- `effective sample ratio ${ratio2.toFixed(2)} below ${options.minEffectiveSampleRatio}`
150
- );
151
- }
152
- if (estimate.maxImportanceWeight > 10) {
153
- reasons.push(`max importance weight ${estimate.maxImportanceWeight.toFixed(2)} is high`);
154
- }
155
- return {
156
- supported: reasons.length === 0,
157
- n: estimate.n,
158
- dropped: options.dropped,
159
- effectiveSampleSize: estimate.effectiveSampleSize,
160
- effectiveSampleRatio: ratio2,
161
- maxImportanceWeight: estimate.maxImportanceWeight,
162
- reasons
163
- };
164
- }
165
- function rewardOf(point) {
166
- if (typeof point.outcome?.reward === "number") return point.outcome.reward;
167
- if (typeof point.outcome?.score === "number") return point.outcome.score;
168
- if (point.outcome?.success === true) return 1;
169
- return 0;
170
- }
171
- function isBehaviorProbability(value) {
172
- return typeof value === "number" && Number.isFinite(value) && value > 0 && value <= 1;
173
- }
174
- function isTargetProbability(value) {
175
- return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
176
- }
177
- function formatProbability(value) {
178
- return typeof value === "number" ? String(value) : String(value ?? "missing");
179
- }
180
- function errorMessage(error) {
181
- return error instanceof Error ? error.message : String(error);
182
- }
183
- function compactDiagnostics(diagnostics, maxDiagnostics) {
184
- if (diagnostics.length <= maxDiagnostics) return diagnostics;
185
- return [
186
- ...diagnostics.slice(0, maxDiagnostics),
187
- `${diagnostics.length - maxDiagnostics} additional OPE diagnostic(s) omitted`
188
- ];
189
- }
190
-
191
- // src/belief-state/selective.ts
192
- var DEFAULT_UTILITY = {
193
- successUtility: 1,
194
- failureUtility: -1,
195
- deferUtility: 0,
196
- verifyCost: 0.05,
197
- askCost: 0.05,
198
- retryCost: 0.1,
199
- stopUtility: 0,
200
- costWeight: 1
201
- };
202
- function thresholdSelectivePolicy(options) {
203
- const threshold = options.confidenceThreshold;
204
- if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) {
205
- throw new ValidationError(
206
- `thresholdSelectivePolicy: confidenceThreshold must be in [0, 1], got ${threshold}`
207
- );
208
- }
209
- const belowThresholdAction = options.belowThresholdAction ?? "verify";
210
- return {
211
- id: options.id ?? `confidence>=${threshold}`,
212
- decide(point) {
213
- const confidence = point.confidence ?? 0;
214
- return {
215
- action: confidence >= threshold ? "accept" : belowThresholdAction,
216
- confidence,
217
- targetProb: point.targetProb,
218
- qHat: point.qHat,
219
- reason: confidence >= threshold ? "confidence threshold passed" : "confidence threshold failed"
220
- };
221
- }
222
- };
223
- }
224
- function evaluateBeliefSelectivePolicy(points, policy, options = {}) {
225
- const utility = { ...DEFAULT_UTILITY, ...options.utility ?? {} };
226
- const scored = points.filter((point) => point.outcome);
227
- const minN = options.minN ?? 30;
228
- const minAccepted = options.minAccepted ?? 5;
229
- const minUtilityDelta = options.minUtilityDelta ?? 0;
230
- const deltas = [];
231
- const acceptedRewards = [];
232
- const rejectedRewards = [];
233
- let baselineUtility = 0;
234
- let policyUtility = 0;
235
- let accepted = 0;
236
- let acceptedErrors = 0;
237
- for (const point of scored) {
238
- const baseline = acceptUtility(point, utility);
239
- const decision = policy.decide(point);
240
- const candidate = policyDecisionUtility(point, decision.action, utility);
241
- const reward = rewardOf2(point, utility);
242
- baselineUtility += baseline;
243
- policyUtility += candidate;
244
- deltas.push(candidate - baseline);
245
- if (decision.action === "accept") {
246
- accepted++;
247
- acceptedRewards.push(reward);
248
- if (reward < 0) acceptedErrors++;
249
- } else {
250
- rejectedRewards.push(reward);
251
- }
252
- }
253
- const n = scored.length;
254
- const rejected = Math.max(0, n - accepted);
255
- const ci = confidenceInterval(deltas, 0.95, { seed: options.seed ?? 17 });
256
- const reasons = [];
257
- if (n < minN) reasons.push(`need at least ${minN} scored decisions, got ${n}`);
258
- if (accepted < minAccepted)
259
- reasons.push(`need at least ${minAccepted} accepted decisions, got ${accepted}`);
260
- if (ci.lower <= minUtilityDelta) {
261
- reasons.push(`utility CI lower bound ${ci.lower.toFixed(4)} does not clear ${minUtilityDelta}`);
262
- }
263
- const recommendation = n < minN || accepted < minAccepted ? "need_more_data" : ci.lower > minUtilityDelta ? "ship" : "hold";
264
- return {
265
- policyId: policy.id,
266
- n,
267
- accepted,
268
- rejected,
269
- coverage: n > 0 ? accepted / n : 0,
270
- acceptedErrorRate: accepted > 0 ? acceptedErrors / accepted : 0,
271
- baselineUtility,
272
- policyUtility,
273
- utilityDelta: policyUtility - baselineUtility,
274
- utilityCi95: ci,
275
- rejectedMeanReward: rejectedRewards.length > 0 ? mean(rejectedRewards) : null,
276
- recommendation,
277
- reasons
278
- };
279
- }
280
- function acceptUtility(point, utility) {
281
- return rewardOf2(point, utility) - utility.costWeight * (point.costUsd ?? point.outcome?.costUsd ?? 0);
282
- }
283
- function policyDecisionUtility(point, action, utility) {
284
- if (action === "accept") return acceptUtility(point, utility);
285
- if (action === "verify") return utility.deferUtility - utility.verifyCost;
286
- if (action === "ask") return utility.deferUtility - utility.askCost;
287
- if (action === "retry") return utility.deferUtility - utility.retryCost;
288
- if (action === "stop") return utility.stopUtility;
289
- return utility.deferUtility;
290
- }
291
- function rewardOf2(point, utility) {
292
- const outcome = point.outcome;
293
- if (!outcome) return utility.failureUtility;
294
- if (typeof outcome.reward === "number") return 2 * outcome.reward - 1;
295
- if (typeof outcome.score === "number") return 2 * outcome.score - 1;
296
- if (outcome.success === true) return utility.successUtility;
297
- if (outcome.success === false) return utility.failureUtility;
298
- return utility.failureUtility;
299
- }
300
- function mean(values) {
301
- return values.reduce((sum, value) => sum + value, 0) / values.length;
302
- }
303
-
304
- // src/belief-state/report.ts
305
- function analyzeBeliefPolicy(options) {
306
- const selective = evaluateBeliefSelectivePolicy(options.points, options.policy, options.selective);
307
- const calibration = calibrateBeliefDecisions(options.points, options.calibration);
308
- const opeTargetPolicy = options.ope?.targetPolicy;
309
- const ope = opeTargetPolicy ? evaluateBeliefOffPolicy(options.points, opeTargetPolicy, options.ope) : null;
310
- const diagnostics = [];
311
- const selectiveStatus = selective.recommendation;
312
- const calibrationStatus = calibration ? "supported" : "unsupported";
313
- const opeRequested = options.requireOpe === true || options.ope !== void 0;
314
- const opeStatus = ope ? ope.support.supported ? "supported" : "unsupported" : opeRequested ? "unsupported" : "not_requested";
315
- if (!calibration) diagnostics.push("calibration unsupported: not enough confidence/outcome pairs");
316
- if (opeRequested && !opeTargetPolicy) diagnostics.push("OPE unsupported: missing target policy");
317
- else if (ope && !ope.support.supported)
318
- diagnostics.push(...ope.support.reasons.map((reason) => `OPE unsupported: ${reason}`));
319
- const status = overallStatus({
320
- selectiveStatus,
321
- hasCalibration: calibration !== null,
322
- opeStatus,
323
- opeRequested
324
- });
325
- return {
326
- policyId: options.policy.id,
327
- n: options.points.length,
328
- status,
329
- selectiveStatus,
330
- calibrationStatus,
331
- opeStatus,
332
- ...ope ? { opeTargetPolicyId: ope.targetPolicyId } : {},
333
- selective,
334
- ...calibration ? { calibration } : {},
335
- ...ope ? { ope } : {},
336
- diagnostics
337
- };
338
- }
339
- function overallStatus(options) {
340
- if (options.selectiveStatus === "need_more_data" || !options.hasCalibration) {
341
- return "need_more_data";
342
- }
343
- if (options.selectiveStatus === "hold") return "hold";
344
- if (options.opeRequested && options.opeStatus !== "supported") return "hold";
345
- return "ship";
346
- }
347
-
348
- // src/belief-state/code-agent-corpus.ts
349
- var FAILURE_RECOVERY_ACTIONS = ["retry", "verify", "continue", "stop"];
350
- var TARGET_LABELS = {
351
- "failure-recovery": "Failure recovery after tool or patch failure",
352
- "tool-selection": "Tool/action selection",
353
- "graph-completion": "Graph completion decision"
354
- };
355
- function extractCodeAgentBeliefDecisionPoints(options) {
356
- const entries = options.entries.filter(isRecord);
357
- const diagnostics = [];
358
- const observed = observedActionsFor(options.source, entries, options);
359
- const decisions = [];
360
- for (const action of observed) {
361
- if (action.kind === "tool" || action.kind === "patch") {
362
- decisions.push(toolSelectionDecision(action, options));
363
- }
364
- if (action.kind === "graph-completion") {
365
- decisions.push(graphCompletionDecision(action, options));
366
- }
367
- }
368
- for (const failed of observed) {
369
- if (failed.kind !== "tool" && failed.kind !== "patch" || failed.success !== false) continue;
370
- const next = observed.find(
371
- (candidate) => candidate.stepIndex > failed.stepIndex && (candidate.kind === "tool" || candidate.kind === "patch" || candidate.kind === "terminal")
372
- );
373
- if (!next) {
374
- diagnostics.push({
375
- runId: options.run.runId,
376
- severity: "warning",
377
- reason: `${failed.id}: failed action has no observable follow-up decision`
378
- });
379
- continue;
380
- }
381
- decisions.push(failureRecoveryDecision(failed, next, options));
382
- }
383
- if (decisions.length === 0) {
384
- diagnostics.push({
385
- runId: options.run.runId,
386
- severity: "info",
387
- reason: `no belief decision points extracted from ${options.source} entries`
388
- });
389
- }
390
- return { decisions, diagnostics };
391
- }
392
- function inventoryBeliefDecisionPoints(points) {
393
- const byKind = [...groupBy(points, (point) => point.kind).entries()].map(
394
- ([kind, bucketPoints]) => bucketFor(kind, bucketPoints, { kind })
395
- ).sort(sortBuckets);
396
- const byTarget = [...groupBy(points, targetIdOf).entries()].filter((entry) => {
397
- return entry[0] !== void 0;
398
- }).map(([targetId, bucketPoints]) => bucketFor(targetId, bucketPoints, { targetId })).sort(sortBuckets);
399
- const diagnostics = [];
400
- if (points.length === 0) diagnostics.push("no decision points available");
401
- for (const bucket of byTarget) {
402
- if (bucket.withOutcome < bucket.n) {
403
- diagnostics.push(`${bucket.id}: ${bucket.n - bucket.withOutcome} decision(s) missing outcome`);
404
- }
405
- if (bucket.withBehaviorProb < bucket.n || bucket.withTargetProb < bucket.n) {
406
- diagnostics.push(`${bucket.id}: OPE support incomplete`);
407
- }
408
- }
409
- return { n: points.length, byKind, byTarget, diagnostics };
410
- }
411
- function selectBeliefDecisionTarget(points, options = {}) {
412
- const minN = options.minN ?? 10;
413
- const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
414
- const preferredTargets = options.preferredTargets ?? [
415
- "failure-recovery",
416
- "tool-selection",
417
- "graph-completion"
418
- ];
419
- const inventory = inventoryBeliefDecisionPoints(points);
420
- for (const targetId of preferredTargets) {
421
- const support = inventory.byTarget.find((bucket) => bucket.targetId === targetId);
422
- if (!support) continue;
423
- const reasons = [];
424
- if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
425
- const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
426
- if (outcomeCoverage < minOutcomeCoverage) {
427
- reasons.push(
428
- `outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
429
- );
430
- }
431
- if (reasons.length > 0) continue;
432
- const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
433
- return {
434
- id: targetId,
435
- label: TARGET_LABELS[targetId],
436
- points: targetPoints,
437
- support,
438
- reasons
439
- };
440
- }
441
- return null;
442
- }
443
- function analyzeBeliefDecisionCorpus(options) {
444
- const inventory = inventoryBeliefDecisionPoints(options.points);
445
- const diagnostics = [...inventory.diagnostics];
446
- const target = options.targetId !== void 0 ? targetSelectionFor(options.points, options.targetId, options) : selectBeliefDecisionTarget(options.points, options);
447
- if (!target) {
448
- diagnostics.push("no decision target has enough support for policy evaluation");
449
- return { inventory, diagnostics };
450
- }
451
- const policy = options.policy ?? thresholdSelectivePolicy({
452
- id: `${target.id}:confidence>=${options.confidenceThreshold ?? 0.5}`,
453
- confidenceThreshold: options.confidenceThreshold ?? 0.5,
454
- belowThresholdAction: "verify"
455
- });
456
- const minN = options.minN ?? 10;
457
- const evaluation = analyzeBeliefPolicy({
458
- points: target.points,
459
- policy,
460
- selective: {
461
- minN,
462
- minAccepted: options.minAccepted ?? Math.min(5, minN),
463
- minUtilityDelta: 0,
464
- ...options.policyOptions?.selective ?? {}
465
- },
466
- calibration: {
467
- minPairs: Math.min(10, minN),
468
- policy,
469
- region: "all",
470
- ...options.policyOptions?.calibration ?? {}
471
- },
472
- ope: {
473
- targetPolicy: embeddedBeliefOpeTargetPolicy(`${target.id}:embedded-target-prob`),
474
- minEffectiveSampleSize: minN,
475
- ...options.policyOptions?.ope ?? {}
476
- },
477
- requireOpe: options.requireOpe ?? true
478
- });
479
- return { inventory, target, policy, evaluation, diagnostics };
480
- }
481
- function observedActionsFor(source, entries, options) {
482
- switch (source) {
483
- case "codex":
484
- return codexObservedActions(entries, options);
485
- case "claude-code":
486
- return claudeObservedActions(entries, options);
487
- case "opencode":
488
- return openCodeObservedActions(entries, options);
489
- case "kimi-code":
490
- return kimiObservedActions(entries, options);
491
- case "pi":
492
- return piObservedActions(entries, options);
493
- }
494
- }
495
- function codexObservedActions(entries, options) {
496
- const actions = [];
497
- const calls = /* @__PURE__ */ new Map();
498
- for (const entry of entries) {
499
- const payload = record(entry.payload) ?? {};
500
- const entryType = stringField(entry, "type");
501
- const payloadType = stringField(payload, "type");
502
- const timestamp = timestampMs(entry.timestamp);
503
- if (entryType === "response_item") {
504
- if (payloadType === "function_call" || payloadType === "custom_tool_call") {
505
- const callId = stringField(payload, "call_id") ?? stringField(payload, "id") ?? `${actions.length}`;
506
- const action = observedAction({
507
- options,
508
- localId: callId,
509
- stepIndex: actions.length,
510
- kind: "tool",
511
- action: stringField(payload, "name") ?? payloadType,
512
- timestamp,
513
- metadata: { sourceEventType: entryType, payloadType }
514
- });
515
- calls.set(callId, action);
516
- actions.push(action);
517
- }
518
- if (payloadType === "function_call_output" || payloadType === "custom_tool_call_output") {
519
- const callId = stringField(payload, "call_id") ?? stringField(payload, "id");
520
- const action = callId ? calls.get(callId) : void 0;
521
- if (action) action.success = !looksLikeError(payload.output);
522
- }
523
- }
524
- if (entryType === "event_msg") {
525
- if (payloadType === "patch_apply_end") {
526
- actions.push(
527
- observedAction({
528
- options,
529
- localId: stringField(payload, "call_id") ?? `patch-${actions.length}`,
530
- stepIndex: actions.length,
531
- kind: "patch",
532
- action: "patch",
533
- timestamp,
534
- success: typeof payload.success === "boolean" ? payload.success : void 0,
535
- metadata: { sourceEventType: entryType, payloadType }
536
- })
537
- );
538
- }
539
- if (payloadType === "task_complete" || payloadType === "turn_aborted") {
540
- actions.push(
541
- observedAction({
542
- options,
543
- localId: `${payloadType}-${actions.length}`,
544
- stepIndex: actions.length,
545
- kind: "terminal",
546
- action: payloadType === "task_complete" ? "stop" : "abort",
547
- timestamp,
548
- success: payloadType === "task_complete",
549
- metadata: { sourceEventType: entryType, payloadType }
550
- })
551
- );
552
- }
553
- }
554
- }
555
- return actions;
556
- }
557
- function claudeObservedActions(entries, options) {
558
- const actions = [];
559
- const calls = /* @__PURE__ */ new Map();
560
- for (const entry of entries) {
561
- const timestamp = timestampMs(entry.timestamp);
562
- const message = record(entry.message);
563
- const content = Array.isArray(message?.content) ? message.content : [];
564
- for (const item of content) {
565
- const part = record(item);
566
- if (!part) continue;
567
- const partType = stringField(part, "type");
568
- if (partType === "tool_use") {
569
- const id = stringField(part, "id") ?? `tool-${actions.length}`;
570
- const action = observedAction({
571
- options,
572
- localId: id,
573
- stepIndex: actions.length,
574
- kind: "tool",
575
- action: stringField(part, "name") ?? "tool",
576
- timestamp,
577
- metadata: { sourceEventType: stringField(entry, "type"), partType }
578
- });
579
- calls.set(id, action);
580
- actions.push(action);
581
- }
582
- if (partType === "tool_result") {
583
- const id = stringField(part, "tool_use_id");
584
- const action = id ? calls.get(id) : void 0;
585
- if (action) action.success = part.is_error !== true;
586
- }
587
- }
588
- if (stringField(entry, "type") === "pr-link") {
589
- actions.push(
590
- observedAction({
591
- options,
592
- localId: `pr-link-${actions.length}`,
593
- stepIndex: actions.length,
594
- kind: "terminal",
595
- action: "stop",
596
- timestamp,
597
- success: true,
598
- metadata: { sourceEventType: "pr-link" }
599
- })
600
- );
601
- }
602
- }
603
- return actions;
604
- }
605
- function openCodeObservedActions(entries, options) {
606
- const actions = [];
607
- for (const entry of entries) {
608
- const type = stringField(entry, "type");
609
- const role = stringField(entry, "role");
610
- const time = record(entry.time);
611
- const timestamp = timestampMs(time?.created);
612
- if (type === "tool") {
613
- const state = record(entry.state);
614
- const status = stringField(state ?? {}, "status");
615
- actions.push(
616
- observedAction({
617
- options,
618
- localId: stringField(entry, "id") ?? `tool-${actions.length}`,
619
- stepIndex: actions.length,
620
- kind: "tool",
621
- action: stringField(entry, "tool") ?? "tool",
622
- timestamp,
623
- success: status === "completed" ? true : status === "error" ? false : void 0,
624
- metadata: { sourceEventType: type, status }
625
- })
626
- );
627
- }
628
- if (type === "patch") {
629
- actions.push(
630
- observedAction({
631
- options,
632
- localId: stringField(entry, "id") ?? `patch-${actions.length}`,
633
- stepIndex: actions.length,
634
- kind: "patch",
635
- action: "patch",
636
- timestamp,
637
- success: true,
638
- metadata: { sourceEventType: type }
639
- })
640
- );
641
- }
642
- if (role === "assistant") {
643
- const finish = stringField(entry, "finish");
644
- if (finish === "stop" || finish === "error") {
645
- actions.push(
646
- observedAction({
647
- options,
648
- localId: stringField(entry, "id") ?? `terminal-${actions.length}`,
649
- stepIndex: actions.length,
650
- kind: "terminal",
651
- action: finish === "stop" ? "stop" : "abort",
652
- timestamp: timestampMs(time?.completed) ?? timestamp,
653
- success: finish === "stop",
654
- costUsd: numberField(entry, "cost"),
655
- metadata: { sourceEventType: "assistant", finish }
656
- })
657
- );
658
- }
659
- }
660
- }
661
- return actions;
662
- }
663
- function kimiObservedActions(entries, options) {
664
- const actions = [];
665
- const calls = /* @__PURE__ */ new Map();
666
- for (const entry of entries) {
667
- const timestamp = timestampMs(entry.timestamp);
668
- const message = record(entry.message);
669
- const messageType = stringField(message ?? {}, "type");
670
- const payload = record(message?.payload) ?? {};
671
- if (messageType === "ToolCall") {
672
- const call = record(payload.function);
673
- const id = stringField(payload, "id") ?? `tool-${actions.length}`;
674
- const action = observedAction({
675
- options,
676
- localId: id,
677
- stepIndex: actions.length,
678
- kind: "tool",
679
- action: stringField(call ?? {}, "name") ?? "tool",
680
- timestamp,
681
- metadata: { sourceEventType: messageType }
682
- });
683
- calls.set(id, action);
684
- actions.push(action);
685
- }
686
- if (messageType === "ToolResult") {
687
- const id = stringField(payload, "tool_call_id");
688
- const action = id ? calls.get(id) : void 0;
689
- if (action) action.success = record(payload.return_value)?.is_error !== true;
690
- }
691
- if (messageType === "TurnEnd" || messageType === "StepInterrupted") {
692
- actions.push(
693
- observedAction({
694
- options,
695
- localId: `${messageType}-${actions.length}`,
696
- stepIndex: actions.length,
697
- kind: "terminal",
698
- action: messageType === "TurnEnd" ? "stop" : "abort",
699
- timestamp,
700
- success: messageType === "TurnEnd",
701
- metadata: { sourceEventType: messageType }
702
- })
703
- );
704
- }
705
- }
706
- return actions;
707
- }
708
- function piObservedActions(entries, options) {
709
- const actions = [];
710
- for (const entry of entries) {
711
- const nodes = Array.isArray(entry.nodes) ? entry.nodes : [];
712
- for (const node of nodes) {
713
- const obj = record(node);
714
- const ir = record(obj?.ir) ?? obj;
715
- const kind = stringField(ir ?? {}, "kind");
716
- if (kind === "ToolInvocation") {
717
- actions.push(
718
- observedAction({
719
- options,
720
- localId: stringField(ir ?? {}, "id") ?? stringField(obj ?? {}, "id") ?? `tool-${actions.length}`,
721
- stepIndex: actions.length,
722
- kind: "tool",
723
- action: "graph-tool",
724
- timestamp: timestampMs(ir?.createdAt),
725
- success: void 0,
726
- metadata: { sourceEventType: "graph-node", graphKind: kind }
727
- })
728
- );
729
- }
730
- if (kind === "ToolResult") {
731
- const prior = [...actions].reverse().find((action) => action.kind === "tool");
732
- if (prior) prior.success = true;
733
- }
734
- if (kind === "CompletionDecision") {
735
- actions.push(
736
- observedAction({
737
- options,
738
- localId: stringField(ir ?? {}, "id") ?? stringField(obj ?? {}, "id") ?? `completion-${actions.length}`,
739
- stepIndex: actions.length,
740
- kind: "graph-completion",
741
- action: "complete",
742
- timestamp: timestampMs(ir?.createdAt),
743
- success: true,
744
- metadata: { sourceEventType: "graph-node", graphKind: kind }
745
- })
746
- );
747
- }
748
- }
749
- }
750
- return actions;
751
- }
752
- function toolSelectionDecision(action, options) {
753
- return {
754
- id: `${options.run.runId}:tool-selection:${action.localId}`,
755
- runId: options.run.runId,
756
- scenarioId: options.run.scenarioId,
757
- stepIndex: action.stepIndex,
758
- kind: "tool-select",
759
- chosenAction: action.action,
760
- candidateActions: [action.action],
761
- confidence: 0.65,
762
- costUsd: action.costUsd,
763
- evidence: action.evidence,
764
- outcome: outcomeFromAction(action, options.run),
765
- metadata: {
766
- target: "tool-selection",
767
- source: options.source,
768
- actionKind: action.kind,
769
- confidenceSource: "fixed-observed-action-prior",
770
- ...action.metadata
771
- }
772
- };
773
- }
774
- function graphCompletionDecision(action, options) {
775
- return {
776
- id: `${options.run.runId}:graph-completion:${action.localId}`,
777
- runId: options.run.runId,
778
- scenarioId: options.run.scenarioId,
779
- stepIndex: action.stepIndex,
780
- kind: "stop",
781
- chosenAction: "complete",
782
- candidateActions: ["complete", "continue", "verify"],
783
- confidence: 0.75,
784
- evidence: action.evidence,
785
- outcome: outcomeFromAction(action, options.run),
786
- metadata: {
787
- target: "graph-completion",
788
- source: options.source,
789
- confidenceSource: "fixed-graph-completion-prior",
790
- ...action.metadata
791
- }
792
- };
793
- }
794
- function failureRecoveryDecision(failed, next, options) {
795
- const chosenAction = classifyFailureRecovery(failed, next);
796
- return {
797
- id: `${options.run.runId}:failure-recovery:${failed.localId}`,
798
- runId: options.run.runId,
799
- scenarioId: options.run.scenarioId,
800
- stepIndex: failed.stepIndex,
801
- kind: "retry",
802
- chosenAction,
803
- candidateActions: [...FAILURE_RECOVERY_ACTIONS],
804
- confidence: recoveryConfidence(chosenAction),
805
- evidence: [...failed.evidence, ...next.evidence],
806
- outcome: outcomeFromAction(next, options.run),
807
- metadata: {
808
- target: "failure-recovery",
809
- source: options.source,
810
- failedActionKind: failed.kind,
811
- failedAction: failed.action,
812
- nextActionKind: next.kind,
813
- nextAction: next.action,
814
- confidenceSource: "heuristic-observed-follow-up"
815
- }
816
- };
817
- }
818
- function classifyFailureRecovery(failed, next) {
819
- if (next.kind === "terminal") return "stop";
820
- if (isVerificationAction(next.action)) return "verify";
821
- if (next.kind === failed.kind && next.action === failed.action) return "retry";
822
- return "continue";
823
- }
824
- function recoveryConfidence(action) {
825
- if (action === "verify") return 0.8;
826
- if (action === "retry") return 0.6;
827
- if (action === "stop") return 0.55;
828
- return 0.35;
829
- }
830
- function isVerificationAction(action) {
831
- const normalized = action.toLowerCase();
832
- return normalized.includes("verify") || normalized.includes("test") || normalized.includes("check") || normalized.includes("lint") || normalized.includes("build") || normalized.includes("typecheck") || normalized.includes("pytest") || normalized.includes("vitest") || normalized.includes("tsc");
833
- }
834
- function outcomeFromAction(action, run) {
835
- const runScore = scoreFromRun(run);
836
- const success = action.success ?? (runScore !== null ? runScore >= 0.5 : void 0);
837
- const score = action.success === void 0 ? runScore ?? void 0 : action.success ? 1 : 0;
838
- if (success === void 0 && score === void 0) return void 0;
839
- return {
840
- ...success !== void 0 ? { success } : {},
841
- ...score !== void 0 ? { score, reward: score } : {},
842
- ...action.costUsd !== void 0 ? { costUsd: action.costUsd } : {},
843
- metadata: {
844
- outcomeSource: action.success === void 0 ? "run-score" : "observed-action-status"
845
- }
846
- };
847
- }
848
- function observedAction(input) {
849
- const id = `${input.options.run.runId}:${input.options.source}:${input.localId}`;
850
- return {
851
- id,
852
- localId: input.localId,
853
- stepIndex: input.stepIndex,
854
- kind: input.kind,
855
- action: input.action,
856
- timestamp: input.timestamp,
857
- success: input.success,
858
- costUsd: input.costUsd,
859
- evidence: [
860
- {
861
- source: "event",
862
- id,
863
- runId: input.options.run.runId,
864
- detail: input.action,
865
- metadata: {
866
- source: input.options.source,
867
- sourcePath: input.options.sourcePath,
868
- ...input.metadata
869
- }
870
- }
871
- ],
872
- metadata: input.metadata ?? {}
873
- };
874
- }
875
- function targetSelectionFor(points, targetId, options) {
876
- const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
877
- if (targetPoints.length === 0) return null;
878
- const support = bucketFor(targetId, targetPoints, { targetId });
879
- const minN = options.minN ?? 10;
880
- const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
881
- const reasons = [];
882
- if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
883
- const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
884
- if (outcomeCoverage < minOutcomeCoverage) {
885
- reasons.push(
886
- `outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
887
- );
888
- }
889
- if (reasons.length > 0) return null;
890
- return { id: targetId, label: TARGET_LABELS[targetId], points: targetPoints, support, reasons };
891
- }
892
- function bucketFor(id, points, identity) {
893
- const outcomes = points.filter((point) => point.outcome);
894
- const scores = outcomes.map((point) => outcomeScore2(point.outcome)).filter((score) => score !== null);
895
- const confidences = points.map((point) => point.confidence).filter((confidence) => typeof confidence === "number");
896
- const successes = outcomes.filter((point) => point.outcome?.success === true).length;
897
- const successDenominator = outcomes.filter(
898
- (point) => typeof point.outcome?.success === "boolean"
899
- ).length;
900
- return {
901
- id,
902
- ...identity,
903
- n: points.length,
904
- withOutcome: outcomes.length,
905
- withConfidence: confidences.length,
906
- withCandidateActions: points.filter((point) => (point.candidateActions?.length ?? 0) > 0).length,
907
- withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
908
- withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
909
- successRate: successDenominator > 0 ? successes / successDenominator : null,
910
- meanScore: scores.length > 0 ? mean2(scores) : null,
911
- meanConfidence: confidences.length > 0 ? mean2(confidences) : null
912
- };
913
- }
914
- function targetIdOf(point) {
915
- const target = point.metadata?.target;
916
- if (target === "failure-recovery" || target === "tool-selection" || target === "graph-completion")
917
- return target;
918
- return void 0;
919
- }
920
- function outcomeScore2(outcome) {
921
- if (!outcome) return null;
922
- if (typeof outcome.score === "number") return outcome.score;
923
- if (typeof outcome.reward === "number") return outcome.reward;
924
- if (outcome.success === true) return 1;
925
- if (outcome.success === false) return 0;
926
- return null;
927
- }
928
- function scoreFromRun(run) {
929
- if (typeof run.outcome.holdoutScore === "number") return run.outcome.holdoutScore;
930
- if (typeof run.outcome.searchScore === "number") return run.outcome.searchScore;
931
- return null;
932
- }
933
- function sortBuckets(a, b) {
934
- return b.n - a.n || a.id.localeCompare(b.id);
935
- }
936
- function groupBy(values, keyOf) {
937
- const map = /* @__PURE__ */ new Map();
938
- for (const value of values) {
939
- const key = keyOf(value);
940
- const bucket = map.get(key);
941
- if (bucket) bucket.push(value);
942
- else map.set(key, [value]);
943
- }
944
- return map;
945
- }
946
- function mean2(values) {
947
- return values.reduce((sum, value) => sum + value, 0) / values.length;
948
- }
949
- function looksLikeError(value) {
950
- if (isRecord(value)) {
951
- if (value.is_error === true || value.error === true) return true;
952
- if ("Err" in value) return true;
953
- }
954
- if (typeof value !== "string") return false;
955
- return /\b(error|failed|exception|traceback)\b/i.test(value);
956
- }
957
- function timestampMs(value) {
958
- if (typeof value === "number" && Number.isFinite(value)) {
959
- return value > 1e12 ? value : value * 1e3;
960
- }
961
- if (typeof value === "string" && value.length > 0) {
962
- const parsed = Date.parse(value);
963
- return Number.isFinite(parsed) ? parsed : void 0;
964
- }
965
- return void 0;
966
- }
967
- function stringField(obj, key) {
968
- const value = obj[key];
969
- return typeof value === "string" && value.length > 0 ? value : void 0;
970
- }
971
- function numberField(obj, key) {
972
- const value = obj[key];
973
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
974
- }
975
- function record(value) {
976
- return isRecord(value) ? value : null;
977
- }
978
- function isRecord(value) {
979
- return value !== null && typeof value === "object" && !Array.isArray(value);
980
- }
981
-
982
- // src/belief-state/research-evidence.ts
983
- function buildBeliefDecisionResearchEvidencePacket(options) {
984
- const claimScope = options.claimScope ?? "counterfactual";
985
- const requireOpe = claimScope === "counterfactual";
986
- const analysis = analyzeBeliefDecisionCorpus({
987
- ...options,
988
- requireOpe: options.requireOpe ?? requireOpe
989
- });
990
- const gates = [
991
- corpusGate(analysis),
992
- selectiveGate(analysis),
993
- calibrationGate(analysis),
994
- ...requireOpe ? [opeGate(analysis)] : []
995
- ];
996
- const caveats = unique([
997
- ...gates.flatMap((gate) => gate.caveats),
998
- ...claimScope === "selective" ? ["counterfactual claims excluded: OPE support was not required"] : []
999
- ]);
1000
- return {
1001
- claimScope,
1002
- status: gates.every((gate) => gate.status === "supported") ? "supported" : "blocked",
1003
- analysis,
1004
- gates,
1005
- blockers: unique(gates.flatMap((gate) => gate.blockers)),
1006
- caveats
1007
- };
1008
- }
1009
- function corpusGate(analysis) {
1010
- const support = analysis.target?.support;
1011
- if (!support) {
1012
- return blocked("corpus", "no decision target has enough outcome support");
1013
- }
1014
- const caveats = support.withBehaviorProb < support.n || support.withTargetProb < support.n ? ["propensity support incomplete; counterfactual claims will require OPE support"] : [];
1015
- return { id: "corpus", status: "supported", blockers: [], caveats };
1016
- }
1017
- function selectiveGate(analysis) {
1018
- const evaluation = analysis.evaluation;
1019
- if (!evaluation) return blocked("selective", "no policy evaluation was produced");
1020
- if (evaluation.selectiveStatus !== "ship") {
1021
- return blocked(
1022
- "selective",
1023
- ...orDefault(
1024
- evaluation.selective.reasons,
1025
- `selective status is ${evaluation.selectiveStatus}`
1026
- )
1027
- );
1028
- }
1029
- return { id: "selective", status: "supported", blockers: [], caveats: [] };
1030
- }
1031
- function calibrationGate(analysis) {
1032
- const evaluation = analysis.evaluation;
1033
- if (!evaluation) return blocked("calibration", "no policy evaluation was produced");
1034
- if (evaluation.calibrationStatus !== "supported") {
1035
- return blocked("calibration", "not enough confidence/outcome pairs for calibration");
1036
- }
1037
- return { id: "calibration", status: "supported", blockers: [], caveats: [] };
1038
- }
1039
- function opeGate(analysis) {
1040
- const evaluation = analysis.evaluation;
1041
- if (!evaluation) return blocked("ope", "no policy evaluation was produced");
1042
- if (evaluation.opeStatus !== "supported") {
1043
- const reasons = evaluation.ope?.support.reasons ?? evaluation.diagnostics.filter((diagnostic) => diagnostic.includes("OPE"));
1044
- return blocked("ope", ...orDefault(reasons, "missing OPE support"));
1045
- }
1046
- return { id: "ope", status: "supported", blockers: [], caveats: [] };
1047
- }
1048
- function blocked(id, ...blockers) {
1049
- return { id, status: "blocked", blockers, caveats: [] };
1050
- }
1051
- function orDefault(values, fallback) {
1052
- return values.length > 0 ? values : [fallback];
1053
- }
1054
- function unique(values) {
1055
- return [...new Set(values)];
1056
- }
1057
-
1058
- // src/belief-state/code-agent-evidence.ts
1059
- function buildCodeAgentBeliefEvidenceCorpus(options) {
1060
- const { sessions, ...evidenceOptions } = options;
1061
- const runs = [];
1062
- const metrics = [];
1063
- const intakeDiagnostics = [];
1064
- const extractionDiagnostics = [];
1065
- const decisions = [];
1066
- for (const session of sessions) {
1067
- const intake = fromCodeAgentBeliefSession(session);
1068
- runs.push(...intake.runs);
1069
- metrics.push(...intake.metrics);
1070
- intakeDiagnostics.push(...intake.diagnostics);
1071
- for (const run of intake.runs) {
1072
- const extraction = extractCodeAgentBeliefDecisionPoints({
1073
- source: session.source,
1074
- entries: session.entries,
1075
- run,
1076
- sourcePath: session.sourcePath
1077
- });
1078
- decisions.push(...extraction.decisions);
1079
- extractionDiagnostics.push(...extraction.diagnostics);
1080
- }
1081
- }
1082
- const evidence = buildBeliefDecisionResearchEvidencePacket({
1083
- ...evidenceOptions,
1084
- points: decisions
1085
- });
1086
- return {
1087
- runs,
1088
- metrics,
1089
- intakeDiagnostics,
1090
- extractionDiagnostics,
1091
- decisions,
1092
- inventory: inventoryBeliefDecisionPoints(decisions),
1093
- evidence
1094
- };
1095
- }
1096
- function fromCodeAgentBeliefSession(session) {
1097
- switch (session.source) {
1098
- case "codex":
1099
- return fromCodexSession(session);
1100
- case "claude-code":
1101
- return fromClaudeCodeSession(session);
1102
- case "opencode":
1103
- return fromOpenCodeSession(session);
1104
- case "kimi-code":
1105
- return fromKimiCodeSession(session);
1106
- case "pi":
1107
- return fromPiSession(session);
1108
- }
1109
- }
1110
-
1111
- // src/belief-state/types.ts
1112
- var BELIEF_DECISION_KINDS = [
1113
- "continue",
1114
- "verify",
1115
- "ask",
1116
- "retry",
1117
- "stop",
1118
- "memory-write",
1119
- "memory-read",
1120
- "tool-select",
1121
- "skill-select",
1122
- "workflow-select",
1123
- "surface-promote"
1124
- ];
1125
- var BELIEF_EVIDENCE_SOURCES = [
1126
- "run",
1127
- "span",
1128
- "event",
1129
- "finding",
1130
- "memory",
1131
- "knowledge",
1132
- "policy"
1133
- ];
1134
- var BELIEF_EVIDENCE_QUALITIES = [
1135
- "direct",
1136
- "derived",
1137
- "self-reported",
1138
- "unverified",
1139
- "stale",
1140
- "contradicted"
1141
- ];
1142
- var BELIEF_EVALUATION_CRITERIA = [
1143
- {
1144
- id: "capture-integrity",
1145
- label: "Capture integrity",
1146
- reasonCodes: ["trace-missing", "run-record-missing", "backend-integrity-missing"]
1147
- },
1148
- {
1149
- id: "decision-completeness",
1150
- label: "Decision completeness",
1151
- reasonCodes: [
1152
- "candidate-actions-missing",
1153
- "chosen-action-missing",
1154
- "decision-evidence-missing"
1155
- ]
1156
- },
1157
- {
1158
- id: "evidence-quality",
1159
- label: "Evidence quality",
1160
- reasonCodes: [
1161
- "evidence-stale",
1162
- "evidence-contradictory",
1163
- "evidence-unverified",
1164
- "evidence-self-reported"
1165
- ]
1166
- },
1167
- {
1168
- id: "outcome-quality",
1169
- label: "Outcome quality",
1170
- reasonCodes: ["outcome-missing", "outcome-delayed", "cost-missing"]
1171
- },
1172
- {
1173
- id: "calibration",
1174
- label: "Calibration",
1175
- reasonCodes: ["confidence-missing", "calibration-unsupported", "calibration-gap-high"]
1176
- },
1177
- {
1178
- id: "accepted-region-risk",
1179
- label: "Accepted-region risk",
1180
- reasonCodes: ["accepted-error-high", "coverage-too-low"]
1181
- },
1182
- {
1183
- id: "policy-value",
1184
- label: "Policy value",
1185
- reasonCodes: ["utility-lift-missing", "baseline-dominates", "cost-too-high"]
1186
- },
1187
- {
1188
- id: "ope-support",
1189
- label: "OPE support",
1190
- reasonCodes: [
1191
- "behavior-propensity-missing",
1192
- "behavior-propensity-invalid",
1193
- "target-propensity-missing",
1194
- "target-propensity-invalid",
1195
- "effective-sample-size-low",
1196
- "importance-weight-high"
1197
- ]
1198
- },
1199
- {
1200
- id: "memory-health",
1201
- label: "Memory health",
1202
- reasonCodes: [
1203
- "memory-stale",
1204
- "memory-poisoning-risk",
1205
- "context-bloat",
1206
- "memory-write-unverified"
1207
- ]
1208
- },
1209
- {
1210
- id: "surface-attribution",
1211
- label: "Surface attribution",
1212
- reasonCodes: ["surface-claim-unsupported", "causal-attribution-missing"]
1213
- },
1214
- {
1215
- id: "generalization",
1216
- label: "Generalization",
1217
- reasonCodes: [
1218
- "split-missing",
1219
- "holdout-regression",
1220
- "task-family-coverage-low",
1221
- "leakage-risk"
1222
- ]
1223
- },
1224
- {
1225
- id: "promotion",
1226
- label: "Promotion",
1227
- reasonCodes: ["negative-control-failed", "promotion-gate-failed", "human-review-required"]
1228
- }
1229
- ];
1230
- function isBeliefDecisionKind(value) {
1231
- return typeof value === "string" && BELIEF_DECISION_KINDS.includes(value);
1232
- }
1233
- function isBeliefEvidenceSource(value) {
1234
- return typeof value === "string" && BELIEF_EVIDENCE_SOURCES.includes(value);
1235
- }
1236
-
1237
- // src/belief-state/extract.ts
1238
- var DECISION_MARKERS = /* @__PURE__ */ new Set(["belief_decision", "belief.decision", "decision_point"]);
1239
- async function extractBeliefDecisionPoints(store, options = {}) {
1240
- const runs = options.runIds ? (await Promise.all(options.runIds.map((runId) => store.getRun(runId)))).filter(Boolean) : await store.listRuns();
1241
- const decisions = [];
1242
- const diagnostics = [];
1243
- for (const run of runs) {
1244
- if (!run) continue;
1245
- const events = await store.events({ runId: run.runId });
1246
- const spans = await store.spans({ runId: run.runId });
1247
- const spanIds = new Set(spans.map((span) => span.spanId));
1248
- let stepIndex = 0;
1249
- for (const event of [...events].sort((a, b) => a.timestamp - b.timestamp)) {
1250
- const parsed = parseDecisionEvent(event, {
1251
- scenarioId: run.scenarioId,
1252
- stepIndex,
1253
- spanExists: event.spanId ? spanIds.has(event.spanId) : false
1254
- });
1255
- if (!parsed) continue;
1256
- if ("diagnostic" in parsed) {
1257
- diagnostics.push(parsed.diagnostic);
1258
- continue;
1259
- }
1260
- decisions.push(parsed.decision);
1261
- stepIndex++;
1262
- }
1263
- }
1264
- return { decisions, diagnostics };
1265
- }
1266
- function parseDecisionEvent(event, context) {
1267
- const payload = event.payload;
1268
- const marker = stringField2(payload, "kind") ?? stringField2(payload, "type");
1269
- if (!marker || !DECISION_MARKERS.has(marker)) return null;
1270
- const decisionKind = stringField2(payload, "decisionKind");
1271
- if (!isBeliefDecisionKind(decisionKind)) {
1272
- return {
1273
- diagnostic: {
1274
- runId: event.runId,
1275
- eventId: event.eventId,
1276
- severity: "warning",
1277
- reason: `belief decision event has unsupported decisionKind "${decisionKind ?? ""}"`
1278
- }
1279
- };
1280
- }
1281
- const chosenAction = stringField2(payload, "chosenAction");
1282
- if (!chosenAction) {
1283
- return {
1284
- diagnostic: {
1285
- runId: event.runId,
1286
- eventId: event.eventId,
1287
- severity: "warning",
1288
- reason: "belief decision event is missing chosenAction"
1289
- }
1290
- };
1291
- }
1292
- const evidence = [
1293
- {
1294
- source: "event",
1295
- id: event.eventId,
1296
- runId: event.runId,
1297
- eventId: event.eventId,
1298
- quality: "direct"
1299
- }
1300
- ];
1301
- if (event.spanId && context.spanExists) {
1302
- evidence.push({
1303
- source: "span",
1304
- id: event.spanId,
1305
- runId: event.runId,
1306
- spanId: event.spanId,
1307
- quality: "direct"
1308
- });
1309
- }
1310
- return {
1311
- decision: {
1312
- id: stringField2(payload, "id") ?? event.eventId,
1313
- runId: event.runId,
1314
- scenarioId: stringField2(payload, "scenarioId") ?? context.scenarioId,
1315
- stepIndex: numberField2(payload, "stepIndex") ?? context.stepIndex,
1316
- kind: decisionKind,
1317
- chosenAction,
1318
- candidateActions: stringArrayField(payload, "candidateActions"),
1319
- confidence: finiteUnitField(payload, "confidence"),
1320
- behaviorProb: numberField2(payload, "behaviorProb"),
1321
- targetProb: numberField2(payload, "targetProb"),
1322
- qHat: finiteUnitField(payload, "qHat"),
1323
- costUsd: nonNegativeNumberField(payload, "costUsd"),
1324
- evidence,
1325
- outcome: parseOutcome(payload),
1326
- metadata: recordField(payload, "metadata")
1327
- }
1328
- };
1329
- }
1330
- function parseOutcome(payload) {
1331
- const value = recordField(payload, "outcome");
1332
- if (!value) return void 0;
1333
- return {
1334
- success: typeof value.success === "boolean" ? value.success : void 0,
1335
- score: finiteUnitField(value, "score"),
1336
- reward: finiteUnitField(value, "reward"),
1337
- costUsd: nonNegativeNumberField(value, "costUsd"),
1338
- observedAt: stringField2(value, "observedAt"),
1339
- metadata: recordField(value, "metadata")
1340
- };
1341
- }
1342
- function stringField2(obj, key) {
1343
- const value = obj[key];
1344
- return typeof value === "string" && value.length > 0 ? value : void 0;
1345
- }
1346
- function numberField2(obj, key) {
1347
- const value = obj[key];
1348
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1349
- }
1350
- function finiteUnitField(obj, key) {
1351
- const value = numberField2(obj, key);
1352
- return value === void 0 ? void 0 : Math.max(0, Math.min(1, value));
1353
- }
1354
- function nonNegativeNumberField(obj, key) {
1355
- const value = numberField2(obj, key);
1356
- return value === void 0 ? void 0 : Math.max(0, value);
1357
- }
1358
- function stringArrayField(obj, key) {
1359
- const value = obj[key];
1360
- if (!Array.isArray(value)) return void 0;
1361
- const strings = value.filter(
1362
- (item) => typeof item === "string" && item.length > 0
1363
- );
1364
- return strings.length > 0 ? strings : void 0;
1365
- }
1366
- function recordField(obj, key) {
1367
- const value = obj[key];
1368
- if (!value || typeof value !== "object" || Array.isArray(value)) return void 0;
1369
- return value;
1370
- }
1371
-
1372
- // src/belief-state/runtime-hooks.ts
1373
- var DEFAULT_MAX_CONTEXT_CHARS = 12e3;
1374
- var DEFAULT_PAYLOAD_PREVIEW_CHARS = 2e3;
1375
- function runtimeDecisionPointToBeliefShadowProbeInput(point, options) {
1376
- const diagnostics = [];
1377
- const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1378
- if (!decisionKind) return { diagnostics };
1379
- const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1380
- const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1381
- return {
1382
- input: {
1383
- probeId: options.probeId,
1384
- decisionId: point.id,
1385
- runId: point.runId,
1386
- scenarioId: point.scenarioId,
1387
- stepIndex: point.stepIndex,
1388
- decisionKind,
1389
- candidateActions: uniqueStrings(point.candidateActions ?? []),
1390
- evidence: evidence.map((ref) => ({
1391
- id: ref.id,
1392
- source: ref.source,
1393
- ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1394
- ...ref.quality ? { quality: ref.quality } : {}
1395
- })),
1396
- context: trimText(point.context, options.maxContextChars),
1397
- metadata: mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence))
1398
- },
1399
- diagnostics
1400
- };
1401
- }
1402
- function runtimeDecisionPointToBeliefDecisionPoint(point, options) {
1403
- const diagnostics = [];
1404
- const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1405
- const chosenAction = stringOrUndefined(options.chosenAction);
1406
- if (!chosenAction) {
1407
- diagnostics.push({
1408
- decisionId: point.id,
1409
- severity: "error",
1410
- reason: "missing chosenAction"
1411
- });
1412
- }
1413
- const candidateActions = uniqueStrings(point.candidateActions ?? []);
1414
- if (chosenAction && candidateActions.length > 0 && !candidateActions.includes(chosenAction)) {
1415
- diagnostics.push({
1416
- decisionId: point.id,
1417
- severity: "warning",
1418
- reason: `chosenAction ${chosenAction} is not in candidateActions`
1419
- });
1420
- }
1421
- if (!decisionKind || !chosenAction) return { diagnostics };
1422
- const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1423
- const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1424
- return {
1425
- point: {
1426
- id: point.id,
1427
- runId: point.runId,
1428
- scenarioId: point.scenarioId,
1429
- stepIndex: point.stepIndex,
1430
- kind: decisionKind,
1431
- chosenAction,
1432
- candidateActions,
1433
- confidence: unitProbabilityOrUndefined(options.confidence),
1434
- behaviorProb: finiteNumberOrUndefined(options.behaviorProb),
1435
- targetProb: finiteNumberOrUndefined(options.targetProb),
1436
- qHat: options.qHat === null ? null : unitProbabilityOrUndefined(options.qHat),
1437
- costUsd: nonNegativeNumberOrUndefined(options.costUsd),
1438
- evidence: evidence.map((ref) => runtimeEvidenceToBeliefEvidence(ref, point)),
1439
- outcome: options.outcome,
1440
- metadata: mergeMetadata(
1441
- mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence)),
1442
- options.metadata
1443
- )
1444
- },
1445
- diagnostics
1446
- };
1447
- }
1448
- function createBeliefRuntimeHookCollector(defaults) {
1449
- const decisions = [];
1450
- const events = [];
1451
- return {
1452
- hooks: {
1453
- onEvent: (event) => {
1454
- events.push(snapshotRuntimeHookEvent(event));
1455
- },
1456
- onDecisionPoint: (point) => {
1457
- decisions.push(snapshotRuntimeDecisionPoint(point));
1458
- }
1459
- },
1460
- decisions,
1461
- events,
1462
- toShadowProbeInputs: (options = {}) => {
1463
- const inputs = [];
1464
- const diagnostics = [];
1465
- const includeLifecycleEvidence = options.includeLifecycleEvidence ?? defaults.includeLifecycleEvidence;
1466
- for (const point of decisions) {
1467
- const report = runtimeDecisionPointToBeliefShadowProbeInput(point, {
1468
- ...defaults,
1469
- ...options,
1470
- includeLifecycleEvidence,
1471
- lifecycleEvents: includeLifecycleEvidence === false ? void 0 : options.lifecycleEvents ?? defaults.lifecycleEvents ?? events
1472
- });
1473
- if (report.input) inputs.push(report.input);
1474
- diagnostics.push(...report.diagnostics);
1475
- }
1476
- return { inputs, diagnostics };
1477
- },
1478
- clear: () => {
1479
- decisions.length = 0;
1480
- events.length = 0;
1481
- }
1482
- };
1483
- }
1484
- function resolveDecisionKind(point, override, diagnostics) {
1485
- const kind = override ?? point.kind;
1486
- if (isBeliefDecisionKind(kind)) return kind;
1487
- diagnostics.push({
1488
- decisionId: point.id,
1489
- severity: "error",
1490
- reason: `unsupported decisionKind "${kind}"`
1491
- });
1492
- return void 0;
1493
- }
1494
- function runtimeEvidenceToBeliefEvidence(ref, point) {
1495
- if (isBeliefEvidenceSource(ref.source)) {
1496
- return {
1497
- source: ref.source,
1498
- id: ref.id,
1499
- runId: point.runId,
1500
- detail: ref.detail,
1501
- quality: ref.quality,
1502
- metadata: ref.metadata
1503
- };
1504
- }
1505
- return {
1506
- source: "event",
1507
- id: ref.id,
1508
- runId: point.runId,
1509
- detail: ref.detail,
1510
- quality: ref.quality,
1511
- metadata: mergeMetadata({ runtimeSource: ref.source }, ref.metadata)
1512
- };
1513
- }
1514
- function runtimeHookEventsToEvidenceRefs(point, options) {
1515
- if (options.includeLifecycleEvidence === false) return [];
1516
- return (options.lifecycleEvents ?? []).filter((event) => runtimeHookEventMatchesDecision(point, event)).map(runtimeHookEventToEvidenceRef);
1517
- }
1518
- function runtimeHookEventMatchesDecision(point, event) {
1519
- if (event.runId !== point.runId) return false;
1520
- if (event.scenarioId && point.scenarioId && event.scenarioId !== point.scenarioId) return false;
1521
- return event.stepIndex === void 0 || event.stepIndex === point.stepIndex;
1522
- }
1523
- function runtimeHookEventToEvidenceRef(event) {
1524
- return {
1525
- source: "runtime_event",
1526
- id: event.id,
1527
- detail: `${event.target}:${event.phase}`,
1528
- quality: "direct",
1529
- metadata: mergeMetadata(
1530
- compactMetadata({
1531
- target: event.target,
1532
- phase: event.phase,
1533
- timestamp: event.timestamp,
1534
- stepIndex: event.stepIndex,
1535
- parentId: event.parentId,
1536
- payloadPreview: previewUnknown(event.payload)
1537
- }),
1538
- event.metadata
1539
- )
1540
- };
1541
- }
1542
- function lifecycleMetadata(refs) {
1543
- if (refs.length === 0) return void 0;
1544
- return {
1545
- lifecycleEventCount: refs.length,
1546
- lifecycleEventIds: refs.map((ref) => ref.id)
1547
- };
1548
- }
1549
- function snapshotRuntimeHookEvent(event) {
1550
- return {
1551
- id: event.id,
1552
- runId: event.runId,
1553
- scenarioId: event.scenarioId,
1554
- target: event.target,
1555
- phase: event.phase,
1556
- timestamp: event.timestamp,
1557
- stepIndex: event.stepIndex,
1558
- parentId: event.parentId,
1559
- payload: snapshotUnknown(event.payload),
1560
- metadata: event.metadata ? { ...event.metadata } : void 0
1561
- };
1562
- }
1563
- function snapshotRuntimeDecisionPoint(point) {
1564
- return {
1565
- id: point.id,
1566
- runId: point.runId,
1567
- scenarioId: point.scenarioId,
1568
- stepIndex: point.stepIndex,
1569
- kind: point.kind,
1570
- candidateActions: [...point.candidateActions ?? []],
1571
- context: point.context,
1572
- evidence: (point.evidence ?? []).map((ref) => ({
1573
- source: ref.source,
1574
- id: ref.id,
1575
- detail: ref.detail,
1576
- quality: ref.quality,
1577
- metadata: ref.metadata ? { ...ref.metadata } : void 0
1578
- })),
1579
- metadata: point.metadata ? { ...point.metadata } : void 0
1580
- };
1581
- }
1582
- function mergeMetadata(base, extra) {
1583
- if (!base && !extra) return void 0;
1584
- return { ...base ?? {}, ...extra ?? {} };
1585
- }
1586
- function compactMetadata(values) {
1587
- const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1588
- return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1589
- }
1590
- function previewUnknown(value, maxChars = DEFAULT_PAYLOAD_PREVIEW_CHARS) {
1591
- if (value === void 0) return void 0;
1592
- if (typeof value === "string") return trimText(value, maxChars);
1593
- try {
1594
- return trimText(JSON.stringify(value), maxChars);
1595
- } catch {
1596
- return trimText(String(value), maxChars);
1597
- }
1598
- }
1599
- function snapshotUnknown(value) {
1600
- if (Array.isArray(value)) return [...value];
1601
- if (isRecord2(value)) return { ...value };
1602
- return value;
1603
- }
1604
- function isRecord2(value) {
1605
- return typeof value === "object" && value !== null && !Array.isArray(value);
1606
- }
1607
- function uniqueStrings(values) {
1608
- return [...new Set(values.filter((value) => value.length > 0))];
1609
- }
1610
- function trimText(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS) {
1611
- if (!value) return void 0;
1612
- return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1613
- }
1614
- function stringOrUndefined(value) {
1615
- return typeof value === "string" && value.length > 0 ? value : void 0;
1616
- }
1617
- function finiteNumberOrUndefined(value) {
1618
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1619
- }
1620
- function unitProbabilityOrUndefined(value) {
1621
- const number = finiteNumberOrUndefined(value);
1622
- return number !== void 0 && number >= 0 && number <= 1 ? number : void 0;
1623
- }
1624
- function nonNegativeNumberOrUndefined(value) {
1625
- const number = finiteNumberOrUndefined(value);
1626
- return number !== void 0 && number >= 0 ? number : void 0;
1627
- }
1628
-
1629
- // src/belief-state/phase0-measurement.ts
1630
- var DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
1631
- function buildRuntimeBeliefPhase0Measurement(options) {
1632
- const runsById = new Map(options.runs.map((run) => [run.runId, run]));
1633
- const labelsByDecisionId = /* @__PURE__ */ new Map();
1634
- const diagnostics = [];
1635
- for (const label of options.labels) {
1636
- if (labelsByDecisionId.has(label.decisionId)) {
1637
- diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
1638
- }
1639
- labelsByDecisionId.set(label.decisionId, label);
1640
- }
1641
- const points = [];
1642
- let missingRunRecordCount = 0;
1643
- let missingLabelCount = 0;
1644
- for (const decision of options.decisions) {
1645
- const run = runsById.get(decision.runId);
1646
- if (!run) {
1647
- missingRunRecordCount += 1;
1648
- diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
1649
- continue;
1650
- }
1651
- const label = labelsByDecisionId.get(decision.id);
1652
- if (!label) {
1653
- missingLabelCount += 1;
1654
- diagnostics.push(`${decision.id}: missing observed action/outcome label`);
1655
- continue;
1656
- }
1657
- const splitTag = label.splitTag ?? run.splitTag;
1658
- const report = runtimeDecisionPointToBeliefDecisionPoint(
1659
- { ...decision, scenarioId: decision.scenarioId ?? run.scenarioId },
1660
- {
1661
- chosenAction: label.chosenAction,
1662
- confidence: label.confidence,
1663
- behaviorProb: label.behaviorProb,
1664
- targetProb: label.targetProb,
1665
- qHat: label.qHat,
1666
- costUsd: label.costUsd,
1667
- outcome: label.outcome,
1668
- lifecycleEvents: options.events,
1669
- metadata: compactMetadata2({
1670
- baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1671
- splitTag,
1672
- ...label.metadata
1673
- })
1674
- }
1675
- );
1676
- diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
1677
- if (report.point) points.push(report.point);
1678
- }
1679
- const packet = buildBeliefDecisionResearchEvidencePacket({
1680
- ...options,
1681
- points
1682
- });
1683
- return {
1684
- points,
1685
- packet,
1686
- summary: summarizePhase0Measurement(options, points, packet, {
1687
- missingRunRecordCount,
1688
- missingLabelCount
1689
- }),
1690
- diagnostics
1691
- };
1692
- }
1693
- function summarizePhase0Measurement(options, points, packet, counts) {
1694
- const producerDecisionCount = options.decisions.length;
1695
- return {
1696
- runCount: options.runs.length,
1697
- producerDecisionCount,
1698
- lifecycleEventCount: options.events?.length ?? 0,
1699
- labelCount: options.labels.length,
1700
- completedPointCount: points.length,
1701
- runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
1702
- labelJoinRate: ratio(points.length, producerDecisionCount),
1703
- missingRunRecordCount: counts.missingRunRecordCount,
1704
- missingLabelCount: counts.missingLabelCount,
1705
- withEvidence: points.filter((point) => point.evidence.length > 0).length,
1706
- withOutcome: points.filter((point) => point.outcome).length,
1707
- withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
1708
- withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
1709
- withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
1710
- baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1711
- packetStatus: packet.status,
1712
- claimScope: packet.claimScope
1713
- };
1714
- }
1715
- function ratio(numerator, denominator) {
1716
- return denominator > 0 ? numerator / denominator : 0;
1717
- }
1718
- function compactMetadata2(values) {
1719
- const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1720
- return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1721
- }
1722
-
1723
- // src/belief-state/runtime-benchmark-corpus.ts
1724
- var MAX_STRING_LENGTH = 12e3;
1725
- var MAX_CONTEXT_LENGTH = 2e4;
1726
- var MAX_EVIDENCE_DETAIL_LENGTH = 2e3;
1727
- var MAX_CANDIDATE_ACTIONS = 50;
1728
- var MAX_EVIDENCE_REFS = 50;
1729
- var MAX_METADATA_DEPTH = 4;
1730
- var MAX_METADATA_KEYS = 100;
1731
- var SENSITIVE_KEY_RE = /(?:authorization|api[_-]?key|token|secret|password|cookie|credential|bearer)/i;
1732
- var SENSITIVE_VALUE_RES = [
1733
- /\bBearer\s+[A-Za-z0-9._~+/=-]+/gi,
1734
- /\b(?:sk|gh[pousr])_[A-Za-z0-9_]{20,}\b/g,
1735
- /\b(?:sk|ghp|gho|ghu|ghs|ghr)-[A-Za-z0-9_-]{20,}\b/g
1736
- ];
1737
- var SENSITIVE_ASSIGNMENT_RE = /\b(api[_-]?key|token|secret|password|cookie)\s*[:=]\s*["']?[^"'\s,;}]+/gi;
1738
- function buildRuntimeBenchmarkBeliefPhase0Measurement(options) {
1739
- const diagnostics = [];
1740
- const trajectory = projectRuntimeTrajectoryEvidence({
1741
- records: options.records,
1742
- defaultSplitTag: options.defaultSplitTag,
1743
- recordIdOf: runtimeBenchmarkRecordId,
1744
- scenarioIdOf: runtimeBenchmarkScenarioId
1745
- });
1746
- const decisions = options.decisions ?? runtimeBenchmarkDecisionPoints(options.records, diagnostics);
1747
- const labels = options.labels ?? [];
1748
- if (decisions.length === 0) {
1749
- diagnostics.push(
1750
- "no runtime decision points supplied or found on records; benchmark lifecycle events alone cannot produce belief decision rows"
1751
- );
1752
- }
1753
- if (labels.length === 0 && decisions.length > 0) {
1754
- diagnostics.push(
1755
- "no decision labels supplied; observed action/outcome joins will be incomplete"
1756
- );
1757
- }
1758
- const measurement = buildRuntimeBeliefPhase0Measurement({
1759
- ...options,
1760
- runs: trajectory.runs,
1761
- events: trajectory.events,
1762
- decisions,
1763
- labels
1764
- });
1765
- return {
1766
- runs: trajectory.runs,
1767
- events: trajectory.events,
1768
- decisions,
1769
- labels,
1770
- trajectory,
1771
- measurement,
1772
- summary: {
1773
- decisionCount: decisions.length,
1774
- labelCount: labels.length
1775
- },
1776
- diagnostics: [...trajectory.diagnostics, ...diagnostics, ...measurement.diagnostics]
1777
- };
1778
- }
1779
- function runtimeBenchmarkRecordId(record2) {
1780
- const parts = [
1781
- nonEmptyString(record2.benchmark),
1782
- nonEmptyString(record2.instanceId),
1783
- nonEmptyString(record2.condition)
1784
- ].filter((part) => part !== void 0);
1785
- return parts.length > 0 ? parts.join(":") : void 0;
1786
- }
1787
- function runtimeBenchmarkScenarioId(record2) {
1788
- return nonEmptyString(record2.instanceId);
1789
- }
1790
- function runtimeBenchmarkDecisionPoints(records, diagnostics) {
1791
- const decisions = [];
1792
- for (let recordIndex = 0; recordIndex < records.length; recordIndex += 1) {
1793
- const record2 = records[recordIndex];
1794
- const raw = record2.runtimeDecisionPoints;
1795
- if (raw === void 0) continue;
1796
- const recordId = runtimeBenchmarkRecordId(record2) ?? `record[${recordIndex}]`;
1797
- if (!Array.isArray(raw)) {
1798
- diagnostics.push(`${recordId}: runtimeDecisionPoints is not an array`);
1799
- continue;
1800
- }
1801
- for (let pointIndex = 0; pointIndex < raw.length; pointIndex += 1) {
1802
- const point = runtimeBenchmarkDecisionPoint(raw[pointIndex], {
1803
- diagnostics,
1804
- path: `${recordId}: runtimeDecisionPoints[${pointIndex}]`
1805
- });
1806
- if (!point) {
1807
- diagnostics.push(
1808
- `${recordId}: runtimeDecisionPoints[${pointIndex}] is not a RuntimeDecisionPoint`
1809
- );
1810
- continue;
1811
- }
1812
- decisions.push(point);
1813
- }
1814
- }
1815
- return decisions;
1816
- }
1817
- function runtimeBenchmarkDecisionPoint(input, context) {
1818
- if (!isRecord3(input)) return null;
1819
- if (typeof input.id !== "string" || input.id.length === 0) return null;
1820
- if (typeof input.runId !== "string" || input.runId.length === 0) return null;
1821
- if (typeof input.stepIndex !== "number" || !Number.isInteger(input.stepIndex) || input.stepIndex < 0) {
1822
- return null;
1823
- }
1824
- if (typeof input.kind !== "string" || input.kind.length === 0) return null;
1825
- return {
1826
- id: sanitizeString(input.id, MAX_STRING_LENGTH),
1827
- runId: sanitizeString(input.runId, MAX_STRING_LENGTH),
1828
- scenarioId: sanitizeOptionalString(input.scenarioId, MAX_STRING_LENGTH),
1829
- stepIndex: input.stepIndex,
1830
- kind: sanitizeString(input.kind, MAX_STRING_LENGTH),
1831
- candidateActions: stringArray(input.candidateActions, {
1832
- ...context,
1833
- maxItems: MAX_CANDIDATE_ACTIONS,
1834
- label: "candidateActions"
1835
- }),
1836
- context: sanitizeOptionalString(input.context, MAX_CONTEXT_LENGTH),
1837
- evidence: runtimeBenchmarkEvidence(input.evidence, context),
1838
- metadata: sanitizeMetadataRecord(input.metadata)
1839
- };
1840
- }
1841
- function runtimeBenchmarkEvidence(input, context) {
1842
- if (!Array.isArray(input)) return [];
1843
- if (input.length > MAX_EVIDENCE_REFS) {
1844
- context.diagnostics.push(`${context.path}: evidence truncated to ${MAX_EVIDENCE_REFS} refs`);
1845
- }
1846
- return input.slice(0, MAX_EVIDENCE_REFS).flatMap((item) => {
1847
- if (!isRecord3(item)) return [];
1848
- const source = sanitizeOptionalString(item.source, MAX_STRING_LENGTH);
1849
- const id = sanitizeOptionalString(item.id, MAX_STRING_LENGTH);
1850
- if (!source || !id) return [];
1851
- return [
1852
- {
1853
- source,
1854
- id,
1855
- detail: sanitizeOptionalString(item.detail, MAX_EVIDENCE_DETAIL_LENGTH),
1856
- metadata: sanitizeMetadataRecord(item.metadata)
1857
- }
1858
- ];
1859
- });
1860
- }
1861
- function stringArray(input, context) {
1862
- if (!Array.isArray(input)) return void 0;
1863
- if (input.length > context.maxItems) {
1864
- context.diagnostics.push(`${context.path}: ${context.label} truncated to ${context.maxItems}`);
1865
- }
1866
- const values = input.slice(0, context.maxItems).filter((value) => typeof value === "string" && value.length > 0).map((value) => sanitizeString(value, MAX_STRING_LENGTH));
1867
- return values.length > 0 ? values : void 0;
1868
- }
1869
- function sanitizeMetadataRecord(metadata) {
1870
- if (!isRecord3(metadata)) return void 0;
1871
- const sanitized = sanitizeMetadata(metadata);
1872
- if (!sanitized || typeof sanitized !== "object" || Array.isArray(sanitized)) return void 0;
1873
- return sanitized;
1874
- }
1875
- function sanitizeMetadata(value, depth = 0) {
1876
- if (value == null) return value;
1877
- if (typeof value === "string") return sanitizeString(value, MAX_STRING_LENGTH);
1878
- if (typeof value === "number" || typeof value === "boolean") return value;
1879
- if (Array.isArray(value)) {
1880
- if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1881
- return value.slice(0, MAX_METADATA_KEYS).map((item) => sanitizeMetadata(item, depth + 1));
1882
- }
1883
- if (!isRecord3(value)) return void 0;
1884
- if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1885
- const sanitized = {};
1886
- for (const [key, nested] of Object.entries(value).slice(0, MAX_METADATA_KEYS)) {
1887
- sanitized[key] = SENSITIVE_KEY_RE.test(key) ? "[REDACTED]" : sanitizeMetadata(nested, depth + 1);
1888
- }
1889
- return sanitized;
1890
- }
1891
- function sanitizeOptionalString(value, maxLength) {
1892
- return typeof value === "string" && value.length > 0 ? sanitizeString(value, maxLength) : void 0;
1893
- }
1894
- function sanitizeString(value, maxLength) {
1895
- let sanitized = value;
1896
- for (const pattern of SENSITIVE_VALUE_RES) {
1897
- sanitized = sanitized.replace(pattern, "[REDACTED]");
1898
- }
1899
- sanitized = sanitized.replace(
1900
- SENSITIVE_ASSIGNMENT_RE,
1901
- (_match, key) => `${key}=[REDACTED]`
1902
- );
1903
- if (sanitized.length <= maxLength) return sanitized;
1904
- return sanitized.slice(0, maxLength);
1905
- }
1906
- function isRecord3(value) {
1907
- return typeof value === "object" && value !== null && !Array.isArray(value);
1908
- }
1909
- function nonEmptyString(value) {
1910
- return typeof value === "string" && value.length > 0 ? value : void 0;
1911
- }
1912
-
1913
- // src/belief-state/shadow-probe.ts
1914
- var DEFAULT_CONCURRENCY = 4;
1915
- var DEFAULT_MAX_CONTEXT_CHARS2 = 12e3;
1916
- async function runBeliefShadowProbe(options) {
1917
- const concurrency = boundedInteger(options.concurrency ?? DEFAULT_CONCURRENCY, 1, 32);
1918
- const records = [];
1919
- const diagnostics = [];
1920
- let next = 0;
1921
- async function worker() {
1922
- while (next < options.points.length) {
1923
- const index = next;
1924
- next += 1;
1925
- const point = options.points[index];
1926
- if (!point) continue;
1927
- const result = await probePoint(point, options);
1928
- records[index] = result.record;
1929
- diagnostics.push(...result.diagnostics);
1930
- }
1931
- }
1932
- await Promise.all(Array.from({ length: Math.min(concurrency, options.points.length) }, worker));
1933
- const completed = records.filter((record2) => !!record2);
1934
- return {
1935
- probeId: options.probeId,
1936
- records: completed,
1937
- diagnostics,
1938
- summary: summarizeShadowProbe(options.points.length, completed)
1939
- };
1940
- }
1941
- function formatBeliefShadowProbePrompt(input) {
1942
- return [
1943
- "Return only JSON. Do not include chain-of-thought.",
1944
- "Infer the agent belief state at this decision boundary using only the context below.",
1945
- "",
1946
- `decisionKind: ${input.decisionKind}`,
1947
- `candidateActions: ${JSON.stringify(input.candidateActions)}`,
1948
- input.observedAction ? `observedAction: ${JSON.stringify(input.observedAction)}` : "",
1949
- input.context ? `context:
1950
- ${input.context}` : "",
1951
- "",
1952
- "Schema:",
1953
- JSON.stringify({
1954
- predictedAction: "one candidate action",
1955
- confidence: "number in [0,1]",
1956
- beliefSummary: "short outcome-blind summary",
1957
- uncertainty: ["short uncertainty"],
1958
- evidenceRefs: ["evidence id"],
1959
- wouldChangeMindIf: ["observable evidence"],
1960
- targetProb: "optional number in [0,1]",
1961
- qHat: "optional number in [0,1]"
1962
- })
1963
- ].filter(Boolean).join("\n");
1964
- }
1965
- async function probePoint(point, options) {
1966
- const diagnostics = [];
1967
- const candidateActions = uniqueStrings2(point.candidateActions ?? []);
1968
- if ((options.requireCandidateActions ?? true) && candidateActions.length === 0) {
1969
- diagnostics.push({
1970
- decisionId: point.id,
1971
- severity: "warning",
1972
- reason: "missing candidateActions"
1973
- });
1974
- return { diagnostics };
1975
- }
1976
- let response;
1977
- try {
1978
- response = await options.probe({
1979
- probeId: options.probeId,
1980
- decisionId: point.id,
1981
- runId: point.runId,
1982
- scenarioId: point.scenarioId,
1983
- stepIndex: point.stepIndex,
1984
- decisionKind: point.kind,
1985
- candidateActions,
1986
- ...options.includeObservedAction ? { observedAction: point.chosenAction } : {},
1987
- evidence: point.evidence.map((ref) => ({
1988
- id: ref.id,
1989
- source: ref.source,
1990
- ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1991
- ...ref.quality ? { quality: ref.quality } : {}
1992
- })),
1993
- context: trimText2(await options.contextOf?.(point), options.maxContextChars),
1994
- metadata: await options.metadataOf?.(point)
1995
- });
1996
- } catch (error) {
1997
- diagnostics.push({
1998
- decisionId: point.id,
1999
- severity: "error",
2000
- reason: `probe threw: ${errorMessage2(error)}`
2001
- });
2002
- return { diagnostics };
2003
- }
2004
- const normalized = normalizeProbeResponse(response, {
2005
- point,
2006
- candidateActions,
2007
- allowOutOfSetActions: options.allowOutOfSetActions ?? false
2008
- });
2009
- if (!normalized.record) {
2010
- diagnostics.push(...normalized.diagnostics);
2011
- return { diagnostics };
2012
- }
2013
- return {
2014
- record: {
2015
- probeId: options.probeId,
2016
- decisionId: point.id,
2017
- runId: point.runId,
2018
- scenarioId: point.scenarioId,
2019
- stepIndex: point.stepIndex,
2020
- decisionKind: point.kind,
2021
- candidateActions,
2022
- observedAction: point.chosenAction,
2023
- agreesWithObservedAction: normalized.record.predictedAction === point.chosenAction,
2024
- ...options.includeOutcomeInRecord === false ? {} : { outcome: point.outcome },
2025
- ...normalized.record
2026
- },
2027
- diagnostics
2028
- };
2029
- }
2030
- function normalizeProbeResponse(response, options) {
2031
- const diagnostics = [];
2032
- const predictedAction = stringOrNull(response.predictedAction);
2033
- if (!predictedAction) {
2034
- diagnostics.push({
2035
- decisionId: options.point.id,
2036
- severity: "error",
2037
- reason: "missing predictedAction"
2038
- });
2039
- } else if (!options.allowOutOfSetActions && options.candidateActions.length > 0 && !options.candidateActions.includes(predictedAction)) {
2040
- diagnostics.push({
2041
- decisionId: options.point.id,
2042
- severity: "error",
2043
- reason: `predictedAction ${predictedAction} is not in candidateActions`
2044
- });
2045
- }
2046
- if (!isUnitProbability(response.confidence)) {
2047
- diagnostics.push({
2048
- decisionId: options.point.id,
2049
- severity: "error",
2050
- reason: `invalid confidence ${String(response.confidence)}`
2051
- });
2052
- }
2053
- if (response.targetProb !== void 0 && !isUnitProbability(response.targetProb)) {
2054
- diagnostics.push({
2055
- decisionId: options.point.id,
2056
- severity: "error",
2057
- reason: `invalid targetProb ${String(response.targetProb)}`
2058
- });
2059
- }
2060
- if (response.qHat !== void 0 && response.qHat !== null && !isUnitProbability(response.qHat)) {
2061
- diagnostics.push({
2062
- decisionId: options.point.id,
2063
- severity: "error",
2064
- reason: `invalid qHat ${String(response.qHat)}`
2065
- });
2066
- }
2067
- if (diagnostics.length > 0 || !predictedAction) return { diagnostics };
2068
- return {
2069
- record: {
2070
- predictedAction,
2071
- confidence: response.confidence,
2072
- ...response.beliefSummary ? { beliefSummary: trimText2(response.beliefSummary, 2e3) } : {},
2073
- uncertainty: compactStrings(response.uncertainty),
2074
- evidenceRefs: compactStrings(response.evidenceRefs),
2075
- wouldChangeMindIf: compactStrings(response.wouldChangeMindIf),
2076
- ...response.targetProb !== void 0 ? { targetProb: response.targetProb } : {},
2077
- ...response.qHat !== void 0 ? { qHat: response.qHat } : {},
2078
- ...response.metadata ? { metadata: response.metadata } : {}
2079
- },
2080
- diagnostics
2081
- };
2082
- }
2083
- function summarizeShadowProbe(attempted, records) {
2084
- const confidences = records.map((record2) => record2.confidence);
2085
- const agreements = records.filter((record2) => record2.agreesWithObservedAction).length;
2086
- return {
2087
- attempted,
2088
- completed: records.length,
2089
- dropped: attempted - records.length,
2090
- withOutcome: records.filter((record2) => record2.outcome !== void 0).length,
2091
- withTargetProb: records.filter((record2) => record2.targetProb !== void 0).length,
2092
- meanConfidence: confidences.length > 0 ? mean3(confidences) : null,
2093
- observedAgreementRate: records.length > 0 ? agreements / records.length : null
2094
- };
2095
- }
2096
- function isUnitProbability(value) {
2097
- return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
2098
- }
2099
- function boundedInteger(value, min, max) {
2100
- if (!Number.isFinite(value)) return min;
2101
- return Math.max(min, Math.min(max, Math.floor(value)));
2102
- }
2103
- function compactStrings(values, maxItems = 12) {
2104
- if (!Array.isArray(values)) return [];
2105
- return values.filter((value) => typeof value === "string" && value.length > 0).slice(0, maxItems).map((value) => trimText2(value, 500) ?? "").filter(Boolean);
2106
- }
2107
- function uniqueStrings2(values) {
2108
- return [...new Set(values.filter((value) => value.length > 0))];
2109
- }
2110
- function stringOrNull(value) {
2111
- return typeof value === "string" && value.length > 0 ? value : null;
2112
- }
2113
- function trimText2(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS2) {
2114
- if (!value) return void 0;
2115
- return value.length > maxChars ? value.slice(value.length - maxChars) : value;
2116
- }
2117
- function mean3(values) {
2118
- return values.reduce((sum, value) => sum + value, 0) / values.length;
2119
- }
2120
- function errorMessage2(error) {
2121
- return error instanceof Error ? error.message : String(error);
2122
- }
2123
- export {
2124
- BELIEF_DECISION_KINDS,
2125
- BELIEF_EVALUATION_CRITERIA,
2126
- BELIEF_EVIDENCE_QUALITIES,
2127
- BELIEF_EVIDENCE_SOURCES,
2128
- analyzeBeliefDecisionCorpus,
2129
- analyzeBeliefPolicy,
2130
- beliefDecisionsToOffPolicyTrajectories,
2131
- buildBeliefDecisionResearchEvidencePacket,
2132
- buildCodeAgentBeliefEvidenceCorpus,
2133
- buildRuntimeBeliefPhase0Measurement,
2134
- buildRuntimeBenchmarkBeliefPhase0Measurement,
2135
- calibrateBeliefDecisions,
2136
- createBeliefRuntimeHookCollector,
2137
- embeddedBeliefOpeTargetPolicy,
2138
- evaluateBeliefOffPolicy,
2139
- evaluateBeliefSelectivePolicy,
2140
- extractBeliefDecisionPoints,
2141
- extractCodeAgentBeliefDecisionPoints,
2142
- formatBeliefShadowProbePrompt,
2143
- inventoryBeliefDecisionPoints,
2144
- isBeliefDecisionKind,
2145
- isBeliefEvidenceSource,
2146
- runBeliefShadowProbe,
2147
- runtimeDecisionPointToBeliefDecisionPoint,
2148
- runtimeDecisionPointToBeliefShadowProbeInput,
2149
- selectBeliefDecisionTarget,
2150
- thresholdSelectivePolicy
2151
- };
2152
- //# sourceMappingURL=index.js.map