@tangle-network/agent-eval 0.120.2 → 0.120.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/analyst/index.d.ts +3111 -0
- package/dist/analyst/index.js +403 -0
- package/dist/analyst/index.js.map +1 -0
- package/dist/authenticity/index.d.ts +161 -0
- package/dist/authenticity/index.js +215 -0
- package/dist/authenticity/index.js.map +1 -0
- package/dist/belief-state/index.d.ts +1301 -0
- package/dist/belief-state/index.js +2152 -0
- package/dist/belief-state/index.js.map +1 -0
- package/dist/benchmarks/index.d.ts +974 -0
- package/dist/benchmarks/index.js +60 -0
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +695 -0
- package/dist/builder-eval/index.js +366 -0
- package/dist/builder-eval/index.js.map +1 -0
- package/dist/campaign/index.d.ts +7454 -0
- package/dist/campaign/index.js +272 -0
- package/dist/campaign/index.js.map +1 -0
- package/dist/chunk-3CDFMEMO.js +3878 -0
- package/dist/chunk-3CDFMEMO.js.map +1 -0
- package/dist/chunk-3RF76KTD.js +84 -0
- package/dist/chunk-3RF76KTD.js.map +1 -0
- package/dist/chunk-3XH4Y2SS.js +750 -0
- package/dist/chunk-3XH4Y2SS.js.map +1 -0
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/chunk-5BYTIDZ7.js +550 -0
- package/dist/chunk-5BYTIDZ7.js.map +1 -0
- package/dist/chunk-5CVUPHJ4.js +2668 -0
- package/dist/chunk-5CVUPHJ4.js.map +1 -0
- package/dist/chunk-ARU2PZFM.js +312 -0
- package/dist/chunk-ARU2PZFM.js.map +1 -0
- package/dist/chunk-BOD4O7OF.js +40 -0
- package/dist/chunk-BOD4O7OF.js.map +1 -0
- package/dist/chunk-CVJP5TMD.js +766 -0
- package/dist/chunk-CVJP5TMD.js.map +1 -0
- package/dist/chunk-DPZAEKA6.js +880 -0
- package/dist/chunk-DPZAEKA6.js.map +1 -0
- package/dist/chunk-DTJ6QUQB.js +131 -0
- package/dist/chunk-DTJ6QUQB.js.map +1 -0
- package/dist/chunk-GGE4NNQT.js +65 -0
- package/dist/chunk-GGE4NNQT.js.map +1 -0
- package/dist/chunk-H5UD2323.js +286 -0
- package/dist/chunk-H5UD2323.js.map +1 -0
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/chunk-HKUCJ437.js +787 -0
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/chunk-JHCHEVET.js +274 -0
- package/dist/chunk-JHCHEVET.js.map +1 -0
- package/dist/chunk-K4DBDHLK.js +158 -0
- package/dist/chunk-K4DBDHLK.js.map +1 -0
- package/dist/chunk-K6N6XJJX.js +306 -0
- package/dist/chunk-K6N6XJJX.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/chunk-MAZ26DC7.js +99 -0
- package/dist/chunk-MAZ26DC7.js.map +1 -0
- package/dist/chunk-MOXWMGPC.js +577 -0
- package/dist/chunk-MOXWMGPC.js.map +1 -0
- package/dist/chunk-NJC7U437.js +626 -0
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/chunk-NMN4WGSJ.js +1030 -0
- package/dist/chunk-NMN4WGSJ.js.map +1 -0
- package/dist/chunk-NPCTHQIO.js +91 -0
- package/dist/chunk-NPCTHQIO.js.map +1 -0
- package/dist/chunk-ONWEPEDO.js +57 -0
- package/dist/chunk-ONWEPEDO.js.map +1 -0
- package/dist/chunk-OYZAPX5G.js +1526 -0
- package/dist/chunk-OYZAPX5G.js.map +1 -0
- package/dist/chunk-P5MGQ2FY.js +7958 -0
- package/dist/chunk-P5MGQ2FY.js.map +1 -0
- package/dist/chunk-PC4UYEBM.js +166 -0
- package/dist/chunk-PC4UYEBM.js.map +1 -0
- package/dist/chunk-PJQFMIOX.js +1182 -0
- package/dist/chunk-PJQFMIOX.js.map +1 -0
- package/dist/chunk-PXD6ZFNY.js +1107 -0
- package/dist/chunk-PXD6ZFNY.js.map +1 -0
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/chunk-PZ5AY32C.js +10 -0
- package/dist/chunk-PZ5AY32C.js.map +1 -0
- package/dist/chunk-QBRSJK47.js +622 -0
- package/dist/chunk-QBRSJK47.js.map +1 -0
- package/dist/chunk-S3UZOQ5Y.js +328 -0
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/chunk-SQQED7ZH.js +998 -0
- package/dist/chunk-SQQED7ZH.js.map +1 -0
- package/dist/chunk-SYV364BL.js +1266 -0
- package/dist/chunk-SYV364BL.js.map +1 -0
- package/dist/chunk-T4SQEITX.js +95 -0
- package/dist/chunk-T4SQEITX.js.map +1 -0
- package/dist/chunk-TT4KNT67.js +124 -0
- package/dist/chunk-TT4KNT67.js.map +1 -0
- package/dist/chunk-U5CHZ5M3.js +357 -0
- package/dist/chunk-U5CHZ5M3.js.map +1 -0
- package/dist/chunk-ULOKLHIQ.js +1937 -0
- package/dist/chunk-ULOKLHIQ.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/chunk-VQMK5FMP.js +247 -0
- package/dist/chunk-VQMK5FMP.js.map +1 -0
- package/dist/chunk-VSMTAMNK.js +53 -0
- package/dist/chunk-VSMTAMNK.js.map +1 -0
- package/dist/chunk-VZSRQ272.js +149 -0
- package/dist/chunk-VZSRQ272.js.map +1 -0
- package/dist/chunk-WW2A73HW.js +159 -0
- package/dist/chunk-WW2A73HW.js.map +1 -0
- package/dist/chunk-X4UCIOTZ.js +136 -0
- package/dist/chunk-X4UCIOTZ.js.map +1 -0
- package/dist/chunk-XJYR7XFV.js +317 -0
- package/dist/chunk-XJYR7XFV.js.map +1 -0
- package/dist/chunk-ZET2UAYW.js +89 -0
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/chunk-ZMXDQ4K7.js +870 -0
- package/dist/chunk-ZMXDQ4K7.js.map +1 -0
- package/dist/chunk-ZZUXHH3R.js +99 -0
- package/dist/chunk-ZZUXHH3R.js.map +1 -0
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +112 -0
- package/dist/cli.js.map +1 -0
- package/dist/contract/index.d.ts +4972 -0
- package/dist/contract/index.js +1654 -0
- package/dist/contract/index.js.map +1 -0
- package/dist/control.d.ts +1013 -0
- package/dist/control.js +34 -0
- package/dist/control.js.map +1 -0
- package/dist/fuzz.d.ts +759 -0
- package/dist/fuzz.js +714 -0
- package/dist/fuzz.js.map +1 -0
- package/dist/hosted/index.d.ts +730 -0
- package/dist/hosted/index.js +14 -0
- package/dist/hosted/index.js.map +1 -0
- package/dist/index.d.ts +16780 -0
- package/dist/index.js +12168 -0
- package/dist/index.js.map +1 -0
- package/dist/matrix/index.d.ts +155 -0
- package/dist/matrix/index.js +8 -0
- package/dist/matrix/index.js.map +1 -0
- package/dist/meta-eval/index.d.ts +1030 -0
- package/dist/meta-eval/index.js +417 -0
- package/dist/meta-eval/index.js.map +1 -0
- package/dist/multishot/index.d.ts +579 -0
- package/dist/multishot/index.js +589 -0
- package/dist/multishot/index.js.map +1 -0
- package/dist/openapi.json +992 -0
- package/dist/pipelines/index.d.ts +567 -0
- package/dist/pipelines/index.js +515 -0
- package/dist/pipelines/index.js.map +1 -0
- package/dist/reporting.d.ts +1277 -0
- package/dist/reporting.js +48 -0
- package/dist/reporting.js.map +1 -0
- package/dist/rl.d.ts +4092 -0
- package/dist/rl.js +1724 -0
- package/dist/rl.js.map +1 -0
- package/dist/run-campaign-75RTPVV5.js +14 -0
- package/dist/run-campaign-75RTPVV5.js.map +1 -0
- package/dist/storyboard/index.d.ts +279 -0
- package/dist/storyboard/index.js +767 -0
- package/dist/storyboard/index.js.map +1 -0
- package/dist/trace-attributes.d.ts +52 -0
- package/dist/trace-attributes.js +62 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +2343 -0
- package/dist/traces.js +249 -0
- package/dist/traces.js.map +1 -0
- package/dist/wire/index.d.ts +1252 -0
- package/dist/wire/index.js +81 -0
- package/dist/wire/index.js.map +1 -0
- package/package.json +1 -1
|
@@ -0,0 +1,2152 @@
|
|
|
1
|
+
import {
|
|
2
|
+
fromClaudeCodeSession,
|
|
3
|
+
fromCodexSession,
|
|
4
|
+
fromKimiCodeSession,
|
|
5
|
+
fromOpenCodeSession,
|
|
6
|
+
fromPiSession
|
|
7
|
+
} from "../chunk-HKUCJ437.js";
|
|
8
|
+
import {
|
|
9
|
+
calibrationFromPairs
|
|
10
|
+
} from "../chunk-NPCTHQIO.js";
|
|
11
|
+
import {
|
|
12
|
+
projectRuntimeTrajectoryEvidence
|
|
13
|
+
} from "../chunk-T4SQEITX.js";
|
|
14
|
+
import {
|
|
15
|
+
offPolicyEstimateAll
|
|
16
|
+
} from "../chunk-DTJ6QUQB.js";
|
|
17
|
+
import {
|
|
18
|
+
confidenceInterval
|
|
19
|
+
} from "../chunk-PJQFMIOX.js";
|
|
20
|
+
import "../chunk-VI2UW6B6.js";
|
|
21
|
+
import "../chunk-PXE2VKMX.js";
|
|
22
|
+
import {
|
|
23
|
+
ValidationError
|
|
24
|
+
} from "../chunk-ONWEPEDO.js";
|
|
25
|
+
import "../chunk-PZ5AY32C.js";
|
|
26
|
+
|
|
27
|
+
// src/belief-state/calibration.ts
|
|
28
|
+
function calibrateBeliefDecisions(points, options = {}) {
|
|
29
|
+
const filtered = filterCalibrationRegion(points, options);
|
|
30
|
+
const pairs = filtered.filter((point) => typeof point.confidence === "number" && point.outcome).map((point) => ({
|
|
31
|
+
evalScore: point.confidence,
|
|
32
|
+
outcome: outcomeScore(point)
|
|
33
|
+
})).filter((pair) => Number.isFinite(pair.outcome));
|
|
34
|
+
const minPairs = options.minPairs ?? 10;
|
|
35
|
+
if (pairs.length < minPairs) return null;
|
|
36
|
+
return calibrationFromPairs(pairs, "belief-confidence", "decision-outcome", {
|
|
37
|
+
bins: options.bins ?? 5,
|
|
38
|
+
range: { lo: 0, hi: 1 }
|
|
39
|
+
});
|
|
40
|
+
}
|
|
41
|
+
function filterCalibrationRegion(points, options) {
|
|
42
|
+
const region = options.region ?? "all";
|
|
43
|
+
if (region === "all") return points;
|
|
44
|
+
const policy = options.policy;
|
|
45
|
+
if (!policy) {
|
|
46
|
+
throw new ValidationError(
|
|
47
|
+
`calibrateBeliefDecisions: policy is required when region is "${region}"`
|
|
48
|
+
);
|
|
49
|
+
}
|
|
50
|
+
return points.filter((point) => {
|
|
51
|
+
const accepted = policy.decide(point).action === "accept";
|
|
52
|
+
return region === "accepted" ? accepted : !accepted;
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
function outcomeScore(point) {
|
|
56
|
+
if (typeof point.outcome?.reward === "number") return point.outcome.reward;
|
|
57
|
+
if (typeof point.outcome?.score === "number") return point.outcome.score;
|
|
58
|
+
if (point.outcome?.success === true) return 1;
|
|
59
|
+
if (point.outcome?.success === false) return 0;
|
|
60
|
+
return Number.NaN;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
// src/belief-state/ope.ts
|
|
64
|
+
function embeddedBeliefOpeTargetPolicy(id = "embedded-target-prob") {
|
|
65
|
+
return {
|
|
66
|
+
id,
|
|
67
|
+
targetProbOf(point) {
|
|
68
|
+
return point.targetProb;
|
|
69
|
+
},
|
|
70
|
+
qHatOf(point) {
|
|
71
|
+
return point.qHat;
|
|
72
|
+
}
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
function beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options = {}) {
|
|
76
|
+
const trajectories = [];
|
|
77
|
+
const diagnostics = [];
|
|
78
|
+
for (const point of points) {
|
|
79
|
+
if (!point.outcome) {
|
|
80
|
+
diagnostics.push(`${point.id}: missing outcome`);
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
if (!isBehaviorProbability(point.behaviorProb)) {
|
|
84
|
+
diagnostics.push(`${point.id}: invalid behaviorProb ${formatProbability(point.behaviorProb)}`);
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
let targetProb;
|
|
88
|
+
let qHat;
|
|
89
|
+
try {
|
|
90
|
+
targetProb = targetPolicy.targetProbOf(point);
|
|
91
|
+
qHat = targetPolicy.qHatOf?.(point);
|
|
92
|
+
} catch (error) {
|
|
93
|
+
diagnostics.push(
|
|
94
|
+
`${point.id}: target policy ${targetPolicy.id} threw (${errorMessage(error)})`
|
|
95
|
+
);
|
|
96
|
+
continue;
|
|
97
|
+
}
|
|
98
|
+
if (!isTargetProbability(targetProb)) {
|
|
99
|
+
diagnostics.push(`${point.id}: invalid targetProb ${formatProbability(targetProb)}`);
|
|
100
|
+
continue;
|
|
101
|
+
}
|
|
102
|
+
if (qHat !== null && qHat !== void 0 && !isTargetProbability(qHat)) {
|
|
103
|
+
diagnostics.push(`${point.id}: invalid qHat ${formatProbability(qHat)}; ignoring qHat`);
|
|
104
|
+
qHat = null;
|
|
105
|
+
}
|
|
106
|
+
trajectories.push({
|
|
107
|
+
runId: point.id,
|
|
108
|
+
reward: rewardOf(point),
|
|
109
|
+
behaviorProb: point.behaviorProb,
|
|
110
|
+
targetProb,
|
|
111
|
+
qHat
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
return {
|
|
115
|
+
targetPolicyId: targetPolicy.id,
|
|
116
|
+
trajectories,
|
|
117
|
+
dropped: points.length - trajectories.length,
|
|
118
|
+
diagnostics: compactDiagnostics(diagnostics, options.maxDiagnostics ?? 20)
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
function evaluateBeliefOffPolicy(points, targetPolicy, options = {}) {
|
|
122
|
+
const trajectoryReport = beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options);
|
|
123
|
+
const { trajectories } = trajectoryReport;
|
|
124
|
+
const estimates = offPolicyEstimateAll(trajectories, options);
|
|
125
|
+
const support = supportDiagnostics(estimates.dr, {
|
|
126
|
+
minEffectiveSampleSize: options.minEffectiveSampleSize ?? 30,
|
|
127
|
+
minEffectiveSampleRatio: options.minEffectiveSampleRatio ?? 0.25,
|
|
128
|
+
dropped: trajectoryReport.dropped,
|
|
129
|
+
diagnostics: trajectoryReport.diagnostics
|
|
130
|
+
});
|
|
131
|
+
return { targetPolicyId: targetPolicy.id, ...estimates, support };
|
|
132
|
+
}
|
|
133
|
+
function supportDiagnostics(estimate, options) {
|
|
134
|
+
const ratio2 = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
|
|
135
|
+
const reasons = [...options.diagnostics];
|
|
136
|
+
if (estimate.n === 0) {
|
|
137
|
+
reasons.push("no valid OPE trajectories");
|
|
138
|
+
}
|
|
139
|
+
if (options.dropped > 0) {
|
|
140
|
+
reasons.push(`dropped ${options.dropped} unsupported decision(s)`);
|
|
141
|
+
}
|
|
142
|
+
if (estimate.effectiveSampleSize < options.minEffectiveSampleSize) {
|
|
143
|
+
reasons.push(
|
|
144
|
+
`effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`
|
|
145
|
+
);
|
|
146
|
+
}
|
|
147
|
+
if (ratio2 < options.minEffectiveSampleRatio) {
|
|
148
|
+
reasons.push(
|
|
149
|
+
`effective sample ratio ${ratio2.toFixed(2)} below ${options.minEffectiveSampleRatio}`
|
|
150
|
+
);
|
|
151
|
+
}
|
|
152
|
+
if (estimate.maxImportanceWeight > 10) {
|
|
153
|
+
reasons.push(`max importance weight ${estimate.maxImportanceWeight.toFixed(2)} is high`);
|
|
154
|
+
}
|
|
155
|
+
return {
|
|
156
|
+
supported: reasons.length === 0,
|
|
157
|
+
n: estimate.n,
|
|
158
|
+
dropped: options.dropped,
|
|
159
|
+
effectiveSampleSize: estimate.effectiveSampleSize,
|
|
160
|
+
effectiveSampleRatio: ratio2,
|
|
161
|
+
maxImportanceWeight: estimate.maxImportanceWeight,
|
|
162
|
+
reasons
|
|
163
|
+
};
|
|
164
|
+
}
|
|
165
|
+
function rewardOf(point) {
|
|
166
|
+
if (typeof point.outcome?.reward === "number") return point.outcome.reward;
|
|
167
|
+
if (typeof point.outcome?.score === "number") return point.outcome.score;
|
|
168
|
+
if (point.outcome?.success === true) return 1;
|
|
169
|
+
return 0;
|
|
170
|
+
}
|
|
171
|
+
function isBehaviorProbability(value) {
|
|
172
|
+
return typeof value === "number" && Number.isFinite(value) && value > 0 && value <= 1;
|
|
173
|
+
}
|
|
174
|
+
function isTargetProbability(value) {
|
|
175
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
|
|
176
|
+
}
|
|
177
|
+
function formatProbability(value) {
|
|
178
|
+
return typeof value === "number" ? String(value) : String(value ?? "missing");
|
|
179
|
+
}
|
|
180
|
+
function errorMessage(error) {
|
|
181
|
+
return error instanceof Error ? error.message : String(error);
|
|
182
|
+
}
|
|
183
|
+
function compactDiagnostics(diagnostics, maxDiagnostics) {
|
|
184
|
+
if (diagnostics.length <= maxDiagnostics) return diagnostics;
|
|
185
|
+
return [
|
|
186
|
+
...diagnostics.slice(0, maxDiagnostics),
|
|
187
|
+
`${diagnostics.length - maxDiagnostics} additional OPE diagnostic(s) omitted`
|
|
188
|
+
];
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
// src/belief-state/selective.ts
|
|
192
|
+
var DEFAULT_UTILITY = {
|
|
193
|
+
successUtility: 1,
|
|
194
|
+
failureUtility: -1,
|
|
195
|
+
deferUtility: 0,
|
|
196
|
+
verifyCost: 0.05,
|
|
197
|
+
askCost: 0.05,
|
|
198
|
+
retryCost: 0.1,
|
|
199
|
+
stopUtility: 0,
|
|
200
|
+
costWeight: 1
|
|
201
|
+
};
|
|
202
|
+
function thresholdSelectivePolicy(options) {
|
|
203
|
+
const threshold = options.confidenceThreshold;
|
|
204
|
+
if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) {
|
|
205
|
+
throw new ValidationError(
|
|
206
|
+
`thresholdSelectivePolicy: confidenceThreshold must be in [0, 1], got ${threshold}`
|
|
207
|
+
);
|
|
208
|
+
}
|
|
209
|
+
const belowThresholdAction = options.belowThresholdAction ?? "verify";
|
|
210
|
+
return {
|
|
211
|
+
id: options.id ?? `confidence>=${threshold}`,
|
|
212
|
+
decide(point) {
|
|
213
|
+
const confidence = point.confidence ?? 0;
|
|
214
|
+
return {
|
|
215
|
+
action: confidence >= threshold ? "accept" : belowThresholdAction,
|
|
216
|
+
confidence,
|
|
217
|
+
targetProb: point.targetProb,
|
|
218
|
+
qHat: point.qHat,
|
|
219
|
+
reason: confidence >= threshold ? "confidence threshold passed" : "confidence threshold failed"
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
};
|
|
223
|
+
}
|
|
224
|
+
function evaluateBeliefSelectivePolicy(points, policy, options = {}) {
|
|
225
|
+
const utility = { ...DEFAULT_UTILITY, ...options.utility ?? {} };
|
|
226
|
+
const scored = points.filter((point) => point.outcome);
|
|
227
|
+
const minN = options.minN ?? 30;
|
|
228
|
+
const minAccepted = options.minAccepted ?? 5;
|
|
229
|
+
const minUtilityDelta = options.minUtilityDelta ?? 0;
|
|
230
|
+
const deltas = [];
|
|
231
|
+
const acceptedRewards = [];
|
|
232
|
+
const rejectedRewards = [];
|
|
233
|
+
let baselineUtility = 0;
|
|
234
|
+
let policyUtility = 0;
|
|
235
|
+
let accepted = 0;
|
|
236
|
+
let acceptedErrors = 0;
|
|
237
|
+
for (const point of scored) {
|
|
238
|
+
const baseline = acceptUtility(point, utility);
|
|
239
|
+
const decision = policy.decide(point);
|
|
240
|
+
const candidate = policyDecisionUtility(point, decision.action, utility);
|
|
241
|
+
const reward = rewardOf2(point, utility);
|
|
242
|
+
baselineUtility += baseline;
|
|
243
|
+
policyUtility += candidate;
|
|
244
|
+
deltas.push(candidate - baseline);
|
|
245
|
+
if (decision.action === "accept") {
|
|
246
|
+
accepted++;
|
|
247
|
+
acceptedRewards.push(reward);
|
|
248
|
+
if (reward < 0) acceptedErrors++;
|
|
249
|
+
} else {
|
|
250
|
+
rejectedRewards.push(reward);
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
const n = scored.length;
|
|
254
|
+
const rejected = Math.max(0, n - accepted);
|
|
255
|
+
const ci = confidenceInterval(deltas, 0.95, { seed: options.seed ?? 17 });
|
|
256
|
+
const reasons = [];
|
|
257
|
+
if (n < minN) reasons.push(`need at least ${minN} scored decisions, got ${n}`);
|
|
258
|
+
if (accepted < minAccepted)
|
|
259
|
+
reasons.push(`need at least ${minAccepted} accepted decisions, got ${accepted}`);
|
|
260
|
+
if (ci.lower <= minUtilityDelta) {
|
|
261
|
+
reasons.push(`utility CI lower bound ${ci.lower.toFixed(4)} does not clear ${minUtilityDelta}`);
|
|
262
|
+
}
|
|
263
|
+
const recommendation = n < minN || accepted < minAccepted ? "need_more_data" : ci.lower > minUtilityDelta ? "ship" : "hold";
|
|
264
|
+
return {
|
|
265
|
+
policyId: policy.id,
|
|
266
|
+
n,
|
|
267
|
+
accepted,
|
|
268
|
+
rejected,
|
|
269
|
+
coverage: n > 0 ? accepted / n : 0,
|
|
270
|
+
acceptedErrorRate: accepted > 0 ? acceptedErrors / accepted : 0,
|
|
271
|
+
baselineUtility,
|
|
272
|
+
policyUtility,
|
|
273
|
+
utilityDelta: policyUtility - baselineUtility,
|
|
274
|
+
utilityCi95: ci,
|
|
275
|
+
rejectedMeanReward: rejectedRewards.length > 0 ? mean(rejectedRewards) : null,
|
|
276
|
+
recommendation,
|
|
277
|
+
reasons
|
|
278
|
+
};
|
|
279
|
+
}
|
|
280
|
+
function acceptUtility(point, utility) {
|
|
281
|
+
return rewardOf2(point, utility) - utility.costWeight * (point.costUsd ?? point.outcome?.costUsd ?? 0);
|
|
282
|
+
}
|
|
283
|
+
function policyDecisionUtility(point, action, utility) {
|
|
284
|
+
if (action === "accept") return acceptUtility(point, utility);
|
|
285
|
+
if (action === "verify") return utility.deferUtility - utility.verifyCost;
|
|
286
|
+
if (action === "ask") return utility.deferUtility - utility.askCost;
|
|
287
|
+
if (action === "retry") return utility.deferUtility - utility.retryCost;
|
|
288
|
+
if (action === "stop") return utility.stopUtility;
|
|
289
|
+
return utility.deferUtility;
|
|
290
|
+
}
|
|
291
|
+
function rewardOf2(point, utility) {
|
|
292
|
+
const outcome = point.outcome;
|
|
293
|
+
if (!outcome) return utility.failureUtility;
|
|
294
|
+
if (typeof outcome.reward === "number") return 2 * outcome.reward - 1;
|
|
295
|
+
if (typeof outcome.score === "number") return 2 * outcome.score - 1;
|
|
296
|
+
if (outcome.success === true) return utility.successUtility;
|
|
297
|
+
if (outcome.success === false) return utility.failureUtility;
|
|
298
|
+
return utility.failureUtility;
|
|
299
|
+
}
|
|
300
|
+
function mean(values) {
|
|
301
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// src/belief-state/report.ts
|
|
305
|
+
function analyzeBeliefPolicy(options) {
|
|
306
|
+
const selective = evaluateBeliefSelectivePolicy(options.points, options.policy, options.selective);
|
|
307
|
+
const calibration = calibrateBeliefDecisions(options.points, options.calibration);
|
|
308
|
+
const opeTargetPolicy = options.ope?.targetPolicy;
|
|
309
|
+
const ope = opeTargetPolicy ? evaluateBeliefOffPolicy(options.points, opeTargetPolicy, options.ope) : null;
|
|
310
|
+
const diagnostics = [];
|
|
311
|
+
const selectiveStatus = selective.recommendation;
|
|
312
|
+
const calibrationStatus = calibration ? "supported" : "unsupported";
|
|
313
|
+
const opeRequested = options.requireOpe === true || options.ope !== void 0;
|
|
314
|
+
const opeStatus = ope ? ope.support.supported ? "supported" : "unsupported" : opeRequested ? "unsupported" : "not_requested";
|
|
315
|
+
if (!calibration) diagnostics.push("calibration unsupported: not enough confidence/outcome pairs");
|
|
316
|
+
if (opeRequested && !opeTargetPolicy) diagnostics.push("OPE unsupported: missing target policy");
|
|
317
|
+
else if (ope && !ope.support.supported)
|
|
318
|
+
diagnostics.push(...ope.support.reasons.map((reason) => `OPE unsupported: ${reason}`));
|
|
319
|
+
const status = overallStatus({
|
|
320
|
+
selectiveStatus,
|
|
321
|
+
hasCalibration: calibration !== null,
|
|
322
|
+
opeStatus,
|
|
323
|
+
opeRequested
|
|
324
|
+
});
|
|
325
|
+
return {
|
|
326
|
+
policyId: options.policy.id,
|
|
327
|
+
n: options.points.length,
|
|
328
|
+
status,
|
|
329
|
+
selectiveStatus,
|
|
330
|
+
calibrationStatus,
|
|
331
|
+
opeStatus,
|
|
332
|
+
...ope ? { opeTargetPolicyId: ope.targetPolicyId } : {},
|
|
333
|
+
selective,
|
|
334
|
+
...calibration ? { calibration } : {},
|
|
335
|
+
...ope ? { ope } : {},
|
|
336
|
+
diagnostics
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
function overallStatus(options) {
|
|
340
|
+
if (options.selectiveStatus === "need_more_data" || !options.hasCalibration) {
|
|
341
|
+
return "need_more_data";
|
|
342
|
+
}
|
|
343
|
+
if (options.selectiveStatus === "hold") return "hold";
|
|
344
|
+
if (options.opeRequested && options.opeStatus !== "supported") return "hold";
|
|
345
|
+
return "ship";
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
// src/belief-state/code-agent-corpus.ts
|
|
349
|
+
var FAILURE_RECOVERY_ACTIONS = ["retry", "verify", "continue", "stop"];
|
|
350
|
+
var TARGET_LABELS = {
|
|
351
|
+
"failure-recovery": "Failure recovery after tool or patch failure",
|
|
352
|
+
"tool-selection": "Tool/action selection",
|
|
353
|
+
"graph-completion": "Graph completion decision"
|
|
354
|
+
};
|
|
355
|
+
function extractCodeAgentBeliefDecisionPoints(options) {
|
|
356
|
+
const entries = options.entries.filter(isRecord);
|
|
357
|
+
const diagnostics = [];
|
|
358
|
+
const observed = observedActionsFor(options.source, entries, options);
|
|
359
|
+
const decisions = [];
|
|
360
|
+
for (const action of observed) {
|
|
361
|
+
if (action.kind === "tool" || action.kind === "patch") {
|
|
362
|
+
decisions.push(toolSelectionDecision(action, options));
|
|
363
|
+
}
|
|
364
|
+
if (action.kind === "graph-completion") {
|
|
365
|
+
decisions.push(graphCompletionDecision(action, options));
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
for (const failed of observed) {
|
|
369
|
+
if (failed.kind !== "tool" && failed.kind !== "patch" || failed.success !== false) continue;
|
|
370
|
+
const next = observed.find(
|
|
371
|
+
(candidate) => candidate.stepIndex > failed.stepIndex && (candidate.kind === "tool" || candidate.kind === "patch" || candidate.kind === "terminal")
|
|
372
|
+
);
|
|
373
|
+
if (!next) {
|
|
374
|
+
diagnostics.push({
|
|
375
|
+
runId: options.run.runId,
|
|
376
|
+
severity: "warning",
|
|
377
|
+
reason: `${failed.id}: failed action has no observable follow-up decision`
|
|
378
|
+
});
|
|
379
|
+
continue;
|
|
380
|
+
}
|
|
381
|
+
decisions.push(failureRecoveryDecision(failed, next, options));
|
|
382
|
+
}
|
|
383
|
+
if (decisions.length === 0) {
|
|
384
|
+
diagnostics.push({
|
|
385
|
+
runId: options.run.runId,
|
|
386
|
+
severity: "info",
|
|
387
|
+
reason: `no belief decision points extracted from ${options.source} entries`
|
|
388
|
+
});
|
|
389
|
+
}
|
|
390
|
+
return { decisions, diagnostics };
|
|
391
|
+
}
|
|
392
|
+
function inventoryBeliefDecisionPoints(points) {
|
|
393
|
+
const byKind = [...groupBy(points, (point) => point.kind).entries()].map(
|
|
394
|
+
([kind, bucketPoints]) => bucketFor(kind, bucketPoints, { kind })
|
|
395
|
+
).sort(sortBuckets);
|
|
396
|
+
const byTarget = [...groupBy(points, targetIdOf).entries()].filter((entry) => {
|
|
397
|
+
return entry[0] !== void 0;
|
|
398
|
+
}).map(([targetId, bucketPoints]) => bucketFor(targetId, bucketPoints, { targetId })).sort(sortBuckets);
|
|
399
|
+
const diagnostics = [];
|
|
400
|
+
if (points.length === 0) diagnostics.push("no decision points available");
|
|
401
|
+
for (const bucket of byTarget) {
|
|
402
|
+
if (bucket.withOutcome < bucket.n) {
|
|
403
|
+
diagnostics.push(`${bucket.id}: ${bucket.n - bucket.withOutcome} decision(s) missing outcome`);
|
|
404
|
+
}
|
|
405
|
+
if (bucket.withBehaviorProb < bucket.n || bucket.withTargetProb < bucket.n) {
|
|
406
|
+
diagnostics.push(`${bucket.id}: OPE support incomplete`);
|
|
407
|
+
}
|
|
408
|
+
}
|
|
409
|
+
return { n: points.length, byKind, byTarget, diagnostics };
|
|
410
|
+
}
|
|
411
|
+
function selectBeliefDecisionTarget(points, options = {}) {
|
|
412
|
+
const minN = options.minN ?? 10;
|
|
413
|
+
const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
|
|
414
|
+
const preferredTargets = options.preferredTargets ?? [
|
|
415
|
+
"failure-recovery",
|
|
416
|
+
"tool-selection",
|
|
417
|
+
"graph-completion"
|
|
418
|
+
];
|
|
419
|
+
const inventory = inventoryBeliefDecisionPoints(points);
|
|
420
|
+
for (const targetId of preferredTargets) {
|
|
421
|
+
const support = inventory.byTarget.find((bucket) => bucket.targetId === targetId);
|
|
422
|
+
if (!support) continue;
|
|
423
|
+
const reasons = [];
|
|
424
|
+
if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
|
|
425
|
+
const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
|
|
426
|
+
if (outcomeCoverage < minOutcomeCoverage) {
|
|
427
|
+
reasons.push(
|
|
428
|
+
`outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
|
|
429
|
+
);
|
|
430
|
+
}
|
|
431
|
+
if (reasons.length > 0) continue;
|
|
432
|
+
const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
|
|
433
|
+
return {
|
|
434
|
+
id: targetId,
|
|
435
|
+
label: TARGET_LABELS[targetId],
|
|
436
|
+
points: targetPoints,
|
|
437
|
+
support,
|
|
438
|
+
reasons
|
|
439
|
+
};
|
|
440
|
+
}
|
|
441
|
+
return null;
|
|
442
|
+
}
|
|
443
|
+
function analyzeBeliefDecisionCorpus(options) {
|
|
444
|
+
const inventory = inventoryBeliefDecisionPoints(options.points);
|
|
445
|
+
const diagnostics = [...inventory.diagnostics];
|
|
446
|
+
const target = options.targetId !== void 0 ? targetSelectionFor(options.points, options.targetId, options) : selectBeliefDecisionTarget(options.points, options);
|
|
447
|
+
if (!target) {
|
|
448
|
+
diagnostics.push("no decision target has enough support for policy evaluation");
|
|
449
|
+
return { inventory, diagnostics };
|
|
450
|
+
}
|
|
451
|
+
const policy = options.policy ?? thresholdSelectivePolicy({
|
|
452
|
+
id: `${target.id}:confidence>=${options.confidenceThreshold ?? 0.5}`,
|
|
453
|
+
confidenceThreshold: options.confidenceThreshold ?? 0.5,
|
|
454
|
+
belowThresholdAction: "verify"
|
|
455
|
+
});
|
|
456
|
+
const minN = options.minN ?? 10;
|
|
457
|
+
const evaluation = analyzeBeliefPolicy({
|
|
458
|
+
points: target.points,
|
|
459
|
+
policy,
|
|
460
|
+
selective: {
|
|
461
|
+
minN,
|
|
462
|
+
minAccepted: options.minAccepted ?? Math.min(5, minN),
|
|
463
|
+
minUtilityDelta: 0,
|
|
464
|
+
...options.policyOptions?.selective ?? {}
|
|
465
|
+
},
|
|
466
|
+
calibration: {
|
|
467
|
+
minPairs: Math.min(10, minN),
|
|
468
|
+
policy,
|
|
469
|
+
region: "all",
|
|
470
|
+
...options.policyOptions?.calibration ?? {}
|
|
471
|
+
},
|
|
472
|
+
ope: {
|
|
473
|
+
targetPolicy: embeddedBeliefOpeTargetPolicy(`${target.id}:embedded-target-prob`),
|
|
474
|
+
minEffectiveSampleSize: minN,
|
|
475
|
+
...options.policyOptions?.ope ?? {}
|
|
476
|
+
},
|
|
477
|
+
requireOpe: options.requireOpe ?? true
|
|
478
|
+
});
|
|
479
|
+
return { inventory, target, policy, evaluation, diagnostics };
|
|
480
|
+
}
|
|
481
|
+
function observedActionsFor(source, entries, options) {
|
|
482
|
+
switch (source) {
|
|
483
|
+
case "codex":
|
|
484
|
+
return codexObservedActions(entries, options);
|
|
485
|
+
case "claude-code":
|
|
486
|
+
return claudeObservedActions(entries, options);
|
|
487
|
+
case "opencode":
|
|
488
|
+
return openCodeObservedActions(entries, options);
|
|
489
|
+
case "kimi-code":
|
|
490
|
+
return kimiObservedActions(entries, options);
|
|
491
|
+
case "pi":
|
|
492
|
+
return piObservedActions(entries, options);
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
function codexObservedActions(entries, options) {
|
|
496
|
+
const actions = [];
|
|
497
|
+
const calls = /* @__PURE__ */ new Map();
|
|
498
|
+
for (const entry of entries) {
|
|
499
|
+
const payload = record(entry.payload) ?? {};
|
|
500
|
+
const entryType = stringField(entry, "type");
|
|
501
|
+
const payloadType = stringField(payload, "type");
|
|
502
|
+
const timestamp = timestampMs(entry.timestamp);
|
|
503
|
+
if (entryType === "response_item") {
|
|
504
|
+
if (payloadType === "function_call" || payloadType === "custom_tool_call") {
|
|
505
|
+
const callId = stringField(payload, "call_id") ?? stringField(payload, "id") ?? `${actions.length}`;
|
|
506
|
+
const action = observedAction({
|
|
507
|
+
options,
|
|
508
|
+
localId: callId,
|
|
509
|
+
stepIndex: actions.length,
|
|
510
|
+
kind: "tool",
|
|
511
|
+
action: stringField(payload, "name") ?? payloadType,
|
|
512
|
+
timestamp,
|
|
513
|
+
metadata: { sourceEventType: entryType, payloadType }
|
|
514
|
+
});
|
|
515
|
+
calls.set(callId, action);
|
|
516
|
+
actions.push(action);
|
|
517
|
+
}
|
|
518
|
+
if (payloadType === "function_call_output" || payloadType === "custom_tool_call_output") {
|
|
519
|
+
const callId = stringField(payload, "call_id") ?? stringField(payload, "id");
|
|
520
|
+
const action = callId ? calls.get(callId) : void 0;
|
|
521
|
+
if (action) action.success = !looksLikeError(payload.output);
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
if (entryType === "event_msg") {
|
|
525
|
+
if (payloadType === "patch_apply_end") {
|
|
526
|
+
actions.push(
|
|
527
|
+
observedAction({
|
|
528
|
+
options,
|
|
529
|
+
localId: stringField(payload, "call_id") ?? `patch-${actions.length}`,
|
|
530
|
+
stepIndex: actions.length,
|
|
531
|
+
kind: "patch",
|
|
532
|
+
action: "patch",
|
|
533
|
+
timestamp,
|
|
534
|
+
success: typeof payload.success === "boolean" ? payload.success : void 0,
|
|
535
|
+
metadata: { sourceEventType: entryType, payloadType }
|
|
536
|
+
})
|
|
537
|
+
);
|
|
538
|
+
}
|
|
539
|
+
if (payloadType === "task_complete" || payloadType === "turn_aborted") {
|
|
540
|
+
actions.push(
|
|
541
|
+
observedAction({
|
|
542
|
+
options,
|
|
543
|
+
localId: `${payloadType}-${actions.length}`,
|
|
544
|
+
stepIndex: actions.length,
|
|
545
|
+
kind: "terminal",
|
|
546
|
+
action: payloadType === "task_complete" ? "stop" : "abort",
|
|
547
|
+
timestamp,
|
|
548
|
+
success: payloadType === "task_complete",
|
|
549
|
+
metadata: { sourceEventType: entryType, payloadType }
|
|
550
|
+
})
|
|
551
|
+
);
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
return actions;
|
|
556
|
+
}
|
|
557
|
+
function claudeObservedActions(entries, options) {
|
|
558
|
+
const actions = [];
|
|
559
|
+
const calls = /* @__PURE__ */ new Map();
|
|
560
|
+
for (const entry of entries) {
|
|
561
|
+
const timestamp = timestampMs(entry.timestamp);
|
|
562
|
+
const message = record(entry.message);
|
|
563
|
+
const content = Array.isArray(message?.content) ? message.content : [];
|
|
564
|
+
for (const item of content) {
|
|
565
|
+
const part = record(item);
|
|
566
|
+
if (!part) continue;
|
|
567
|
+
const partType = stringField(part, "type");
|
|
568
|
+
if (partType === "tool_use") {
|
|
569
|
+
const id = stringField(part, "id") ?? `tool-${actions.length}`;
|
|
570
|
+
const action = observedAction({
|
|
571
|
+
options,
|
|
572
|
+
localId: id,
|
|
573
|
+
stepIndex: actions.length,
|
|
574
|
+
kind: "tool",
|
|
575
|
+
action: stringField(part, "name") ?? "tool",
|
|
576
|
+
timestamp,
|
|
577
|
+
metadata: { sourceEventType: stringField(entry, "type"), partType }
|
|
578
|
+
});
|
|
579
|
+
calls.set(id, action);
|
|
580
|
+
actions.push(action);
|
|
581
|
+
}
|
|
582
|
+
if (partType === "tool_result") {
|
|
583
|
+
const id = stringField(part, "tool_use_id");
|
|
584
|
+
const action = id ? calls.get(id) : void 0;
|
|
585
|
+
if (action) action.success = part.is_error !== true;
|
|
586
|
+
}
|
|
587
|
+
}
|
|
588
|
+
if (stringField(entry, "type") === "pr-link") {
|
|
589
|
+
actions.push(
|
|
590
|
+
observedAction({
|
|
591
|
+
options,
|
|
592
|
+
localId: `pr-link-${actions.length}`,
|
|
593
|
+
stepIndex: actions.length,
|
|
594
|
+
kind: "terminal",
|
|
595
|
+
action: "stop",
|
|
596
|
+
timestamp,
|
|
597
|
+
success: true,
|
|
598
|
+
metadata: { sourceEventType: "pr-link" }
|
|
599
|
+
})
|
|
600
|
+
);
|
|
601
|
+
}
|
|
602
|
+
}
|
|
603
|
+
return actions;
|
|
604
|
+
}
|
|
605
|
+
function openCodeObservedActions(entries, options) {
|
|
606
|
+
const actions = [];
|
|
607
|
+
for (const entry of entries) {
|
|
608
|
+
const type = stringField(entry, "type");
|
|
609
|
+
const role = stringField(entry, "role");
|
|
610
|
+
const time = record(entry.time);
|
|
611
|
+
const timestamp = timestampMs(time?.created);
|
|
612
|
+
if (type === "tool") {
|
|
613
|
+
const state = record(entry.state);
|
|
614
|
+
const status = stringField(state ?? {}, "status");
|
|
615
|
+
actions.push(
|
|
616
|
+
observedAction({
|
|
617
|
+
options,
|
|
618
|
+
localId: stringField(entry, "id") ?? `tool-${actions.length}`,
|
|
619
|
+
stepIndex: actions.length,
|
|
620
|
+
kind: "tool",
|
|
621
|
+
action: stringField(entry, "tool") ?? "tool",
|
|
622
|
+
timestamp,
|
|
623
|
+
success: status === "completed" ? true : status === "error" ? false : void 0,
|
|
624
|
+
metadata: { sourceEventType: type, status }
|
|
625
|
+
})
|
|
626
|
+
);
|
|
627
|
+
}
|
|
628
|
+
if (type === "patch") {
|
|
629
|
+
actions.push(
|
|
630
|
+
observedAction({
|
|
631
|
+
options,
|
|
632
|
+
localId: stringField(entry, "id") ?? `patch-${actions.length}`,
|
|
633
|
+
stepIndex: actions.length,
|
|
634
|
+
kind: "patch",
|
|
635
|
+
action: "patch",
|
|
636
|
+
timestamp,
|
|
637
|
+
success: true,
|
|
638
|
+
metadata: { sourceEventType: type }
|
|
639
|
+
})
|
|
640
|
+
);
|
|
641
|
+
}
|
|
642
|
+
if (role === "assistant") {
|
|
643
|
+
const finish = stringField(entry, "finish");
|
|
644
|
+
if (finish === "stop" || finish === "error") {
|
|
645
|
+
actions.push(
|
|
646
|
+
observedAction({
|
|
647
|
+
options,
|
|
648
|
+
localId: stringField(entry, "id") ?? `terminal-${actions.length}`,
|
|
649
|
+
stepIndex: actions.length,
|
|
650
|
+
kind: "terminal",
|
|
651
|
+
action: finish === "stop" ? "stop" : "abort",
|
|
652
|
+
timestamp: timestampMs(time?.completed) ?? timestamp,
|
|
653
|
+
success: finish === "stop",
|
|
654
|
+
costUsd: numberField(entry, "cost"),
|
|
655
|
+
metadata: { sourceEventType: "assistant", finish }
|
|
656
|
+
})
|
|
657
|
+
);
|
|
658
|
+
}
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
return actions;
|
|
662
|
+
}
|
|
663
|
+
function kimiObservedActions(entries, options) {
|
|
664
|
+
const actions = [];
|
|
665
|
+
const calls = /* @__PURE__ */ new Map();
|
|
666
|
+
for (const entry of entries) {
|
|
667
|
+
const timestamp = timestampMs(entry.timestamp);
|
|
668
|
+
const message = record(entry.message);
|
|
669
|
+
const messageType = stringField(message ?? {}, "type");
|
|
670
|
+
const payload = record(message?.payload) ?? {};
|
|
671
|
+
if (messageType === "ToolCall") {
|
|
672
|
+
const call = record(payload.function);
|
|
673
|
+
const id = stringField(payload, "id") ?? `tool-${actions.length}`;
|
|
674
|
+
const action = observedAction({
|
|
675
|
+
options,
|
|
676
|
+
localId: id,
|
|
677
|
+
stepIndex: actions.length,
|
|
678
|
+
kind: "tool",
|
|
679
|
+
action: stringField(call ?? {}, "name") ?? "tool",
|
|
680
|
+
timestamp,
|
|
681
|
+
metadata: { sourceEventType: messageType }
|
|
682
|
+
});
|
|
683
|
+
calls.set(id, action);
|
|
684
|
+
actions.push(action);
|
|
685
|
+
}
|
|
686
|
+
if (messageType === "ToolResult") {
|
|
687
|
+
const id = stringField(payload, "tool_call_id");
|
|
688
|
+
const action = id ? calls.get(id) : void 0;
|
|
689
|
+
if (action) action.success = record(payload.return_value)?.is_error !== true;
|
|
690
|
+
}
|
|
691
|
+
if (messageType === "TurnEnd" || messageType === "StepInterrupted") {
|
|
692
|
+
actions.push(
|
|
693
|
+
observedAction({
|
|
694
|
+
options,
|
|
695
|
+
localId: `${messageType}-${actions.length}`,
|
|
696
|
+
stepIndex: actions.length,
|
|
697
|
+
kind: "terminal",
|
|
698
|
+
action: messageType === "TurnEnd" ? "stop" : "abort",
|
|
699
|
+
timestamp,
|
|
700
|
+
success: messageType === "TurnEnd",
|
|
701
|
+
metadata: { sourceEventType: messageType }
|
|
702
|
+
})
|
|
703
|
+
);
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
return actions;
|
|
707
|
+
}
|
|
708
|
+
function piObservedActions(entries, options) {
|
|
709
|
+
const actions = [];
|
|
710
|
+
for (const entry of entries) {
|
|
711
|
+
const nodes = Array.isArray(entry.nodes) ? entry.nodes : [];
|
|
712
|
+
for (const node of nodes) {
|
|
713
|
+
const obj = record(node);
|
|
714
|
+
const ir = record(obj?.ir) ?? obj;
|
|
715
|
+
const kind = stringField(ir ?? {}, "kind");
|
|
716
|
+
if (kind === "ToolInvocation") {
|
|
717
|
+
actions.push(
|
|
718
|
+
observedAction({
|
|
719
|
+
options,
|
|
720
|
+
localId: stringField(ir ?? {}, "id") ?? stringField(obj ?? {}, "id") ?? `tool-${actions.length}`,
|
|
721
|
+
stepIndex: actions.length,
|
|
722
|
+
kind: "tool",
|
|
723
|
+
action: "graph-tool",
|
|
724
|
+
timestamp: timestampMs(ir?.createdAt),
|
|
725
|
+
success: void 0,
|
|
726
|
+
metadata: { sourceEventType: "graph-node", graphKind: kind }
|
|
727
|
+
})
|
|
728
|
+
);
|
|
729
|
+
}
|
|
730
|
+
if (kind === "ToolResult") {
|
|
731
|
+
const prior = [...actions].reverse().find((action) => action.kind === "tool");
|
|
732
|
+
if (prior) prior.success = true;
|
|
733
|
+
}
|
|
734
|
+
if (kind === "CompletionDecision") {
|
|
735
|
+
actions.push(
|
|
736
|
+
observedAction({
|
|
737
|
+
options,
|
|
738
|
+
localId: stringField(ir ?? {}, "id") ?? stringField(obj ?? {}, "id") ?? `completion-${actions.length}`,
|
|
739
|
+
stepIndex: actions.length,
|
|
740
|
+
kind: "graph-completion",
|
|
741
|
+
action: "complete",
|
|
742
|
+
timestamp: timestampMs(ir?.createdAt),
|
|
743
|
+
success: true,
|
|
744
|
+
metadata: { sourceEventType: "graph-node", graphKind: kind }
|
|
745
|
+
})
|
|
746
|
+
);
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
}
|
|
750
|
+
return actions;
|
|
751
|
+
}
|
|
752
|
+
function toolSelectionDecision(action, options) {
|
|
753
|
+
return {
|
|
754
|
+
id: `${options.run.runId}:tool-selection:${action.localId}`,
|
|
755
|
+
runId: options.run.runId,
|
|
756
|
+
scenarioId: options.run.scenarioId,
|
|
757
|
+
stepIndex: action.stepIndex,
|
|
758
|
+
kind: "tool-select",
|
|
759
|
+
chosenAction: action.action,
|
|
760
|
+
candidateActions: [action.action],
|
|
761
|
+
confidence: 0.65,
|
|
762
|
+
costUsd: action.costUsd,
|
|
763
|
+
evidence: action.evidence,
|
|
764
|
+
outcome: outcomeFromAction(action, options.run),
|
|
765
|
+
metadata: {
|
|
766
|
+
target: "tool-selection",
|
|
767
|
+
source: options.source,
|
|
768
|
+
actionKind: action.kind,
|
|
769
|
+
confidenceSource: "fixed-observed-action-prior",
|
|
770
|
+
...action.metadata
|
|
771
|
+
}
|
|
772
|
+
};
|
|
773
|
+
}
|
|
774
|
+
function graphCompletionDecision(action, options) {
|
|
775
|
+
return {
|
|
776
|
+
id: `${options.run.runId}:graph-completion:${action.localId}`,
|
|
777
|
+
runId: options.run.runId,
|
|
778
|
+
scenarioId: options.run.scenarioId,
|
|
779
|
+
stepIndex: action.stepIndex,
|
|
780
|
+
kind: "stop",
|
|
781
|
+
chosenAction: "complete",
|
|
782
|
+
candidateActions: ["complete", "continue", "verify"],
|
|
783
|
+
confidence: 0.75,
|
|
784
|
+
evidence: action.evidence,
|
|
785
|
+
outcome: outcomeFromAction(action, options.run),
|
|
786
|
+
metadata: {
|
|
787
|
+
target: "graph-completion",
|
|
788
|
+
source: options.source,
|
|
789
|
+
confidenceSource: "fixed-graph-completion-prior",
|
|
790
|
+
...action.metadata
|
|
791
|
+
}
|
|
792
|
+
};
|
|
793
|
+
}
|
|
794
|
+
function failureRecoveryDecision(failed, next, options) {
|
|
795
|
+
const chosenAction = classifyFailureRecovery(failed, next);
|
|
796
|
+
return {
|
|
797
|
+
id: `${options.run.runId}:failure-recovery:${failed.localId}`,
|
|
798
|
+
runId: options.run.runId,
|
|
799
|
+
scenarioId: options.run.scenarioId,
|
|
800
|
+
stepIndex: failed.stepIndex,
|
|
801
|
+
kind: "retry",
|
|
802
|
+
chosenAction,
|
|
803
|
+
candidateActions: [...FAILURE_RECOVERY_ACTIONS],
|
|
804
|
+
confidence: recoveryConfidence(chosenAction),
|
|
805
|
+
evidence: [...failed.evidence, ...next.evidence],
|
|
806
|
+
outcome: outcomeFromAction(next, options.run),
|
|
807
|
+
metadata: {
|
|
808
|
+
target: "failure-recovery",
|
|
809
|
+
source: options.source,
|
|
810
|
+
failedActionKind: failed.kind,
|
|
811
|
+
failedAction: failed.action,
|
|
812
|
+
nextActionKind: next.kind,
|
|
813
|
+
nextAction: next.action,
|
|
814
|
+
confidenceSource: "heuristic-observed-follow-up"
|
|
815
|
+
}
|
|
816
|
+
};
|
|
817
|
+
}
|
|
818
|
+
function classifyFailureRecovery(failed, next) {
|
|
819
|
+
if (next.kind === "terminal") return "stop";
|
|
820
|
+
if (isVerificationAction(next.action)) return "verify";
|
|
821
|
+
if (next.kind === failed.kind && next.action === failed.action) return "retry";
|
|
822
|
+
return "continue";
|
|
823
|
+
}
|
|
824
|
+
function recoveryConfidence(action) {
|
|
825
|
+
if (action === "verify") return 0.8;
|
|
826
|
+
if (action === "retry") return 0.6;
|
|
827
|
+
if (action === "stop") return 0.55;
|
|
828
|
+
return 0.35;
|
|
829
|
+
}
|
|
830
|
+
function isVerificationAction(action) {
|
|
831
|
+
const normalized = action.toLowerCase();
|
|
832
|
+
return normalized.includes("verify") || normalized.includes("test") || normalized.includes("check") || normalized.includes("lint") || normalized.includes("build") || normalized.includes("typecheck") || normalized.includes("pytest") || normalized.includes("vitest") || normalized.includes("tsc");
|
|
833
|
+
}
|
|
834
|
+
function outcomeFromAction(action, run) {
|
|
835
|
+
const runScore = scoreFromRun(run);
|
|
836
|
+
const success = action.success ?? (runScore !== null ? runScore >= 0.5 : void 0);
|
|
837
|
+
const score = action.success === void 0 ? runScore ?? void 0 : action.success ? 1 : 0;
|
|
838
|
+
if (success === void 0 && score === void 0) return void 0;
|
|
839
|
+
return {
|
|
840
|
+
...success !== void 0 ? { success } : {},
|
|
841
|
+
...score !== void 0 ? { score, reward: score } : {},
|
|
842
|
+
...action.costUsd !== void 0 ? { costUsd: action.costUsd } : {},
|
|
843
|
+
metadata: {
|
|
844
|
+
outcomeSource: action.success === void 0 ? "run-score" : "observed-action-status"
|
|
845
|
+
}
|
|
846
|
+
};
|
|
847
|
+
}
|
|
848
|
+
function observedAction(input) {
|
|
849
|
+
const id = `${input.options.run.runId}:${input.options.source}:${input.localId}`;
|
|
850
|
+
return {
|
|
851
|
+
id,
|
|
852
|
+
localId: input.localId,
|
|
853
|
+
stepIndex: input.stepIndex,
|
|
854
|
+
kind: input.kind,
|
|
855
|
+
action: input.action,
|
|
856
|
+
timestamp: input.timestamp,
|
|
857
|
+
success: input.success,
|
|
858
|
+
costUsd: input.costUsd,
|
|
859
|
+
evidence: [
|
|
860
|
+
{
|
|
861
|
+
source: "event",
|
|
862
|
+
id,
|
|
863
|
+
runId: input.options.run.runId,
|
|
864
|
+
detail: input.action,
|
|
865
|
+
metadata: {
|
|
866
|
+
source: input.options.source,
|
|
867
|
+
sourcePath: input.options.sourcePath,
|
|
868
|
+
...input.metadata
|
|
869
|
+
}
|
|
870
|
+
}
|
|
871
|
+
],
|
|
872
|
+
metadata: input.metadata ?? {}
|
|
873
|
+
};
|
|
874
|
+
}
|
|
875
|
+
function targetSelectionFor(points, targetId, options) {
|
|
876
|
+
const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
|
|
877
|
+
if (targetPoints.length === 0) return null;
|
|
878
|
+
const support = bucketFor(targetId, targetPoints, { targetId });
|
|
879
|
+
const minN = options.minN ?? 10;
|
|
880
|
+
const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
|
|
881
|
+
const reasons = [];
|
|
882
|
+
if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
|
|
883
|
+
const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
|
|
884
|
+
if (outcomeCoverage < minOutcomeCoverage) {
|
|
885
|
+
reasons.push(
|
|
886
|
+
`outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
|
|
887
|
+
);
|
|
888
|
+
}
|
|
889
|
+
if (reasons.length > 0) return null;
|
|
890
|
+
return { id: targetId, label: TARGET_LABELS[targetId], points: targetPoints, support, reasons };
|
|
891
|
+
}
|
|
892
|
+
function bucketFor(id, points, identity) {
|
|
893
|
+
const outcomes = points.filter((point) => point.outcome);
|
|
894
|
+
const scores = outcomes.map((point) => outcomeScore2(point.outcome)).filter((score) => score !== null);
|
|
895
|
+
const confidences = points.map((point) => point.confidence).filter((confidence) => typeof confidence === "number");
|
|
896
|
+
const successes = outcomes.filter((point) => point.outcome?.success === true).length;
|
|
897
|
+
const successDenominator = outcomes.filter(
|
|
898
|
+
(point) => typeof point.outcome?.success === "boolean"
|
|
899
|
+
).length;
|
|
900
|
+
return {
|
|
901
|
+
id,
|
|
902
|
+
...identity,
|
|
903
|
+
n: points.length,
|
|
904
|
+
withOutcome: outcomes.length,
|
|
905
|
+
withConfidence: confidences.length,
|
|
906
|
+
withCandidateActions: points.filter((point) => (point.candidateActions?.length ?? 0) > 0).length,
|
|
907
|
+
withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
|
|
908
|
+
withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
|
|
909
|
+
successRate: successDenominator > 0 ? successes / successDenominator : null,
|
|
910
|
+
meanScore: scores.length > 0 ? mean2(scores) : null,
|
|
911
|
+
meanConfidence: confidences.length > 0 ? mean2(confidences) : null
|
|
912
|
+
};
|
|
913
|
+
}
|
|
914
|
+
function targetIdOf(point) {
|
|
915
|
+
const target = point.metadata?.target;
|
|
916
|
+
if (target === "failure-recovery" || target === "tool-selection" || target === "graph-completion")
|
|
917
|
+
return target;
|
|
918
|
+
return void 0;
|
|
919
|
+
}
|
|
920
|
+
function outcomeScore2(outcome) {
|
|
921
|
+
if (!outcome) return null;
|
|
922
|
+
if (typeof outcome.score === "number") return outcome.score;
|
|
923
|
+
if (typeof outcome.reward === "number") return outcome.reward;
|
|
924
|
+
if (outcome.success === true) return 1;
|
|
925
|
+
if (outcome.success === false) return 0;
|
|
926
|
+
return null;
|
|
927
|
+
}
|
|
928
|
+
function scoreFromRun(run) {
|
|
929
|
+
if (typeof run.outcome.holdoutScore === "number") return run.outcome.holdoutScore;
|
|
930
|
+
if (typeof run.outcome.searchScore === "number") return run.outcome.searchScore;
|
|
931
|
+
return null;
|
|
932
|
+
}
|
|
933
|
+
function sortBuckets(a, b) {
|
|
934
|
+
return b.n - a.n || a.id.localeCompare(b.id);
|
|
935
|
+
}
|
|
936
|
+
function groupBy(values, keyOf) {
|
|
937
|
+
const map = /* @__PURE__ */ new Map();
|
|
938
|
+
for (const value of values) {
|
|
939
|
+
const key = keyOf(value);
|
|
940
|
+
const bucket = map.get(key);
|
|
941
|
+
if (bucket) bucket.push(value);
|
|
942
|
+
else map.set(key, [value]);
|
|
943
|
+
}
|
|
944
|
+
return map;
|
|
945
|
+
}
|
|
946
|
+
function mean2(values) {
|
|
947
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
948
|
+
}
|
|
949
|
+
function looksLikeError(value) {
|
|
950
|
+
if (isRecord(value)) {
|
|
951
|
+
if (value.is_error === true || value.error === true) return true;
|
|
952
|
+
if ("Err" in value) return true;
|
|
953
|
+
}
|
|
954
|
+
if (typeof value !== "string") return false;
|
|
955
|
+
return /\b(error|failed|exception|traceback)\b/i.test(value);
|
|
956
|
+
}
|
|
957
|
+
function timestampMs(value) {
|
|
958
|
+
if (typeof value === "number" && Number.isFinite(value)) {
|
|
959
|
+
return value > 1e12 ? value : value * 1e3;
|
|
960
|
+
}
|
|
961
|
+
if (typeof value === "string" && value.length > 0) {
|
|
962
|
+
const parsed = Date.parse(value);
|
|
963
|
+
return Number.isFinite(parsed) ? parsed : void 0;
|
|
964
|
+
}
|
|
965
|
+
return void 0;
|
|
966
|
+
}
|
|
967
|
+
function stringField(obj, key) {
|
|
968
|
+
const value = obj[key];
|
|
969
|
+
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
970
|
+
}
|
|
971
|
+
function numberField(obj, key) {
|
|
972
|
+
const value = obj[key];
|
|
973
|
+
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
974
|
+
}
|
|
975
|
+
function record(value) {
|
|
976
|
+
return isRecord(value) ? value : null;
|
|
977
|
+
}
|
|
978
|
+
function isRecord(value) {
|
|
979
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
980
|
+
}
|
|
981
|
+
|
|
982
|
+
// src/belief-state/research-evidence.ts
|
|
983
|
+
function buildBeliefDecisionResearchEvidencePacket(options) {
|
|
984
|
+
const claimScope = options.claimScope ?? "counterfactual";
|
|
985
|
+
const requireOpe = claimScope === "counterfactual";
|
|
986
|
+
const analysis = analyzeBeliefDecisionCorpus({
|
|
987
|
+
...options,
|
|
988
|
+
requireOpe: options.requireOpe ?? requireOpe
|
|
989
|
+
});
|
|
990
|
+
const gates = [
|
|
991
|
+
corpusGate(analysis),
|
|
992
|
+
selectiveGate(analysis),
|
|
993
|
+
calibrationGate(analysis),
|
|
994
|
+
...requireOpe ? [opeGate(analysis)] : []
|
|
995
|
+
];
|
|
996
|
+
const caveats = unique([
|
|
997
|
+
...gates.flatMap((gate) => gate.caveats),
|
|
998
|
+
...claimScope === "selective" ? ["counterfactual claims excluded: OPE support was not required"] : []
|
|
999
|
+
]);
|
|
1000
|
+
return {
|
|
1001
|
+
claimScope,
|
|
1002
|
+
status: gates.every((gate) => gate.status === "supported") ? "supported" : "blocked",
|
|
1003
|
+
analysis,
|
|
1004
|
+
gates,
|
|
1005
|
+
blockers: unique(gates.flatMap((gate) => gate.blockers)),
|
|
1006
|
+
caveats
|
|
1007
|
+
};
|
|
1008
|
+
}
|
|
1009
|
+
function corpusGate(analysis) {
|
|
1010
|
+
const support = analysis.target?.support;
|
|
1011
|
+
if (!support) {
|
|
1012
|
+
return blocked("corpus", "no decision target has enough outcome support");
|
|
1013
|
+
}
|
|
1014
|
+
const caveats = support.withBehaviorProb < support.n || support.withTargetProb < support.n ? ["propensity support incomplete; counterfactual claims will require OPE support"] : [];
|
|
1015
|
+
return { id: "corpus", status: "supported", blockers: [], caveats };
|
|
1016
|
+
}
|
|
1017
|
+
function selectiveGate(analysis) {
|
|
1018
|
+
const evaluation = analysis.evaluation;
|
|
1019
|
+
if (!evaluation) return blocked("selective", "no policy evaluation was produced");
|
|
1020
|
+
if (evaluation.selectiveStatus !== "ship") {
|
|
1021
|
+
return blocked(
|
|
1022
|
+
"selective",
|
|
1023
|
+
...orDefault(
|
|
1024
|
+
evaluation.selective.reasons,
|
|
1025
|
+
`selective status is ${evaluation.selectiveStatus}`
|
|
1026
|
+
)
|
|
1027
|
+
);
|
|
1028
|
+
}
|
|
1029
|
+
return { id: "selective", status: "supported", blockers: [], caveats: [] };
|
|
1030
|
+
}
|
|
1031
|
+
function calibrationGate(analysis) {
|
|
1032
|
+
const evaluation = analysis.evaluation;
|
|
1033
|
+
if (!evaluation) return blocked("calibration", "no policy evaluation was produced");
|
|
1034
|
+
if (evaluation.calibrationStatus !== "supported") {
|
|
1035
|
+
return blocked("calibration", "not enough confidence/outcome pairs for calibration");
|
|
1036
|
+
}
|
|
1037
|
+
return { id: "calibration", status: "supported", blockers: [], caveats: [] };
|
|
1038
|
+
}
|
|
1039
|
+
function opeGate(analysis) {
|
|
1040
|
+
const evaluation = analysis.evaluation;
|
|
1041
|
+
if (!evaluation) return blocked("ope", "no policy evaluation was produced");
|
|
1042
|
+
if (evaluation.opeStatus !== "supported") {
|
|
1043
|
+
const reasons = evaluation.ope?.support.reasons ?? evaluation.diagnostics.filter((diagnostic) => diagnostic.includes("OPE"));
|
|
1044
|
+
return blocked("ope", ...orDefault(reasons, "missing OPE support"));
|
|
1045
|
+
}
|
|
1046
|
+
return { id: "ope", status: "supported", blockers: [], caveats: [] };
|
|
1047
|
+
}
|
|
1048
|
+
function blocked(id, ...blockers) {
|
|
1049
|
+
return { id, status: "blocked", blockers, caveats: [] };
|
|
1050
|
+
}
|
|
1051
|
+
function orDefault(values, fallback) {
|
|
1052
|
+
return values.length > 0 ? values : [fallback];
|
|
1053
|
+
}
|
|
1054
|
+
function unique(values) {
|
|
1055
|
+
return [...new Set(values)];
|
|
1056
|
+
}
|
|
1057
|
+
|
|
1058
|
+
// src/belief-state/code-agent-evidence.ts
|
|
1059
|
+
function buildCodeAgentBeliefEvidenceCorpus(options) {
|
|
1060
|
+
const { sessions, ...evidenceOptions } = options;
|
|
1061
|
+
const runs = [];
|
|
1062
|
+
const metrics = [];
|
|
1063
|
+
const intakeDiagnostics = [];
|
|
1064
|
+
const extractionDiagnostics = [];
|
|
1065
|
+
const decisions = [];
|
|
1066
|
+
for (const session of sessions) {
|
|
1067
|
+
const intake = fromCodeAgentBeliefSession(session);
|
|
1068
|
+
runs.push(...intake.runs);
|
|
1069
|
+
metrics.push(...intake.metrics);
|
|
1070
|
+
intakeDiagnostics.push(...intake.diagnostics);
|
|
1071
|
+
for (const run of intake.runs) {
|
|
1072
|
+
const extraction = extractCodeAgentBeliefDecisionPoints({
|
|
1073
|
+
source: session.source,
|
|
1074
|
+
entries: session.entries,
|
|
1075
|
+
run,
|
|
1076
|
+
sourcePath: session.sourcePath
|
|
1077
|
+
});
|
|
1078
|
+
decisions.push(...extraction.decisions);
|
|
1079
|
+
extractionDiagnostics.push(...extraction.diagnostics);
|
|
1080
|
+
}
|
|
1081
|
+
}
|
|
1082
|
+
const evidence = buildBeliefDecisionResearchEvidencePacket({
|
|
1083
|
+
...evidenceOptions,
|
|
1084
|
+
points: decisions
|
|
1085
|
+
});
|
|
1086
|
+
return {
|
|
1087
|
+
runs,
|
|
1088
|
+
metrics,
|
|
1089
|
+
intakeDiagnostics,
|
|
1090
|
+
extractionDiagnostics,
|
|
1091
|
+
decisions,
|
|
1092
|
+
inventory: inventoryBeliefDecisionPoints(decisions),
|
|
1093
|
+
evidence
|
|
1094
|
+
};
|
|
1095
|
+
}
|
|
1096
|
+
function fromCodeAgentBeliefSession(session) {
|
|
1097
|
+
switch (session.source) {
|
|
1098
|
+
case "codex":
|
|
1099
|
+
return fromCodexSession(session);
|
|
1100
|
+
case "claude-code":
|
|
1101
|
+
return fromClaudeCodeSession(session);
|
|
1102
|
+
case "opencode":
|
|
1103
|
+
return fromOpenCodeSession(session);
|
|
1104
|
+
case "kimi-code":
|
|
1105
|
+
return fromKimiCodeSession(session);
|
|
1106
|
+
case "pi":
|
|
1107
|
+
return fromPiSession(session);
|
|
1108
|
+
}
|
|
1109
|
+
}
|
|
1110
|
+
|
|
1111
|
+
// src/belief-state/types.ts
|
|
1112
|
+
var BELIEF_DECISION_KINDS = [
|
|
1113
|
+
"continue",
|
|
1114
|
+
"verify",
|
|
1115
|
+
"ask",
|
|
1116
|
+
"retry",
|
|
1117
|
+
"stop",
|
|
1118
|
+
"memory-write",
|
|
1119
|
+
"memory-read",
|
|
1120
|
+
"tool-select",
|
|
1121
|
+
"skill-select",
|
|
1122
|
+
"workflow-select",
|
|
1123
|
+
"surface-promote"
|
|
1124
|
+
];
|
|
1125
|
+
var BELIEF_EVIDENCE_SOURCES = [
|
|
1126
|
+
"run",
|
|
1127
|
+
"span",
|
|
1128
|
+
"event",
|
|
1129
|
+
"finding",
|
|
1130
|
+
"memory",
|
|
1131
|
+
"knowledge",
|
|
1132
|
+
"policy"
|
|
1133
|
+
];
|
|
1134
|
+
var BELIEF_EVIDENCE_QUALITIES = [
|
|
1135
|
+
"direct",
|
|
1136
|
+
"derived",
|
|
1137
|
+
"self-reported",
|
|
1138
|
+
"unverified",
|
|
1139
|
+
"stale",
|
|
1140
|
+
"contradicted"
|
|
1141
|
+
];
|
|
1142
|
+
var BELIEF_EVALUATION_CRITERIA = [
|
|
1143
|
+
{
|
|
1144
|
+
id: "capture-integrity",
|
|
1145
|
+
label: "Capture integrity",
|
|
1146
|
+
reasonCodes: ["trace-missing", "run-record-missing", "backend-integrity-missing"]
|
|
1147
|
+
},
|
|
1148
|
+
{
|
|
1149
|
+
id: "decision-completeness",
|
|
1150
|
+
label: "Decision completeness",
|
|
1151
|
+
reasonCodes: [
|
|
1152
|
+
"candidate-actions-missing",
|
|
1153
|
+
"chosen-action-missing",
|
|
1154
|
+
"decision-evidence-missing"
|
|
1155
|
+
]
|
|
1156
|
+
},
|
|
1157
|
+
{
|
|
1158
|
+
id: "evidence-quality",
|
|
1159
|
+
label: "Evidence quality",
|
|
1160
|
+
reasonCodes: [
|
|
1161
|
+
"evidence-stale",
|
|
1162
|
+
"evidence-contradictory",
|
|
1163
|
+
"evidence-unverified",
|
|
1164
|
+
"evidence-self-reported"
|
|
1165
|
+
]
|
|
1166
|
+
},
|
|
1167
|
+
{
|
|
1168
|
+
id: "outcome-quality",
|
|
1169
|
+
label: "Outcome quality",
|
|
1170
|
+
reasonCodes: ["outcome-missing", "outcome-delayed", "cost-missing"]
|
|
1171
|
+
},
|
|
1172
|
+
{
|
|
1173
|
+
id: "calibration",
|
|
1174
|
+
label: "Calibration",
|
|
1175
|
+
reasonCodes: ["confidence-missing", "calibration-unsupported", "calibration-gap-high"]
|
|
1176
|
+
},
|
|
1177
|
+
{
|
|
1178
|
+
id: "accepted-region-risk",
|
|
1179
|
+
label: "Accepted-region risk",
|
|
1180
|
+
reasonCodes: ["accepted-error-high", "coverage-too-low"]
|
|
1181
|
+
},
|
|
1182
|
+
{
|
|
1183
|
+
id: "policy-value",
|
|
1184
|
+
label: "Policy value",
|
|
1185
|
+
reasonCodes: ["utility-lift-missing", "baseline-dominates", "cost-too-high"]
|
|
1186
|
+
},
|
|
1187
|
+
{
|
|
1188
|
+
id: "ope-support",
|
|
1189
|
+
label: "OPE support",
|
|
1190
|
+
reasonCodes: [
|
|
1191
|
+
"behavior-propensity-missing",
|
|
1192
|
+
"behavior-propensity-invalid",
|
|
1193
|
+
"target-propensity-missing",
|
|
1194
|
+
"target-propensity-invalid",
|
|
1195
|
+
"effective-sample-size-low",
|
|
1196
|
+
"importance-weight-high"
|
|
1197
|
+
]
|
|
1198
|
+
},
|
|
1199
|
+
{
|
|
1200
|
+
id: "memory-health",
|
|
1201
|
+
label: "Memory health",
|
|
1202
|
+
reasonCodes: [
|
|
1203
|
+
"memory-stale",
|
|
1204
|
+
"memory-poisoning-risk",
|
|
1205
|
+
"context-bloat",
|
|
1206
|
+
"memory-write-unverified"
|
|
1207
|
+
]
|
|
1208
|
+
},
|
|
1209
|
+
{
|
|
1210
|
+
id: "surface-attribution",
|
|
1211
|
+
label: "Surface attribution",
|
|
1212
|
+
reasonCodes: ["surface-claim-unsupported", "causal-attribution-missing"]
|
|
1213
|
+
},
|
|
1214
|
+
{
|
|
1215
|
+
id: "generalization",
|
|
1216
|
+
label: "Generalization",
|
|
1217
|
+
reasonCodes: [
|
|
1218
|
+
"split-missing",
|
|
1219
|
+
"holdout-regression",
|
|
1220
|
+
"task-family-coverage-low",
|
|
1221
|
+
"leakage-risk"
|
|
1222
|
+
]
|
|
1223
|
+
},
|
|
1224
|
+
{
|
|
1225
|
+
id: "promotion",
|
|
1226
|
+
label: "Promotion",
|
|
1227
|
+
reasonCodes: ["negative-control-failed", "promotion-gate-failed", "human-review-required"]
|
|
1228
|
+
}
|
|
1229
|
+
];
|
|
1230
|
+
function isBeliefDecisionKind(value) {
|
|
1231
|
+
return typeof value === "string" && BELIEF_DECISION_KINDS.includes(value);
|
|
1232
|
+
}
|
|
1233
|
+
function isBeliefEvidenceSource(value) {
|
|
1234
|
+
return typeof value === "string" && BELIEF_EVIDENCE_SOURCES.includes(value);
|
|
1235
|
+
}
|
|
1236
|
+
|
|
1237
|
+
// src/belief-state/extract.ts
|
|
1238
|
+
var DECISION_MARKERS = /* @__PURE__ */ new Set(["belief_decision", "belief.decision", "decision_point"]);
|
|
1239
|
+
async function extractBeliefDecisionPoints(store, options = {}) {
|
|
1240
|
+
const runs = options.runIds ? (await Promise.all(options.runIds.map((runId) => store.getRun(runId)))).filter(Boolean) : await store.listRuns();
|
|
1241
|
+
const decisions = [];
|
|
1242
|
+
const diagnostics = [];
|
|
1243
|
+
for (const run of runs) {
|
|
1244
|
+
if (!run) continue;
|
|
1245
|
+
const events = await store.events({ runId: run.runId });
|
|
1246
|
+
const spans = await store.spans({ runId: run.runId });
|
|
1247
|
+
const spanIds = new Set(spans.map((span) => span.spanId));
|
|
1248
|
+
let stepIndex = 0;
|
|
1249
|
+
for (const event of [...events].sort((a, b) => a.timestamp - b.timestamp)) {
|
|
1250
|
+
const parsed = parseDecisionEvent(event, {
|
|
1251
|
+
scenarioId: run.scenarioId,
|
|
1252
|
+
stepIndex,
|
|
1253
|
+
spanExists: event.spanId ? spanIds.has(event.spanId) : false
|
|
1254
|
+
});
|
|
1255
|
+
if (!parsed) continue;
|
|
1256
|
+
if ("diagnostic" in parsed) {
|
|
1257
|
+
diagnostics.push(parsed.diagnostic);
|
|
1258
|
+
continue;
|
|
1259
|
+
}
|
|
1260
|
+
decisions.push(parsed.decision);
|
|
1261
|
+
stepIndex++;
|
|
1262
|
+
}
|
|
1263
|
+
}
|
|
1264
|
+
return { decisions, diagnostics };
|
|
1265
|
+
}
|
|
1266
|
+
function parseDecisionEvent(event, context) {
|
|
1267
|
+
const payload = event.payload;
|
|
1268
|
+
const marker = stringField2(payload, "kind") ?? stringField2(payload, "type");
|
|
1269
|
+
if (!marker || !DECISION_MARKERS.has(marker)) return null;
|
|
1270
|
+
const decisionKind = stringField2(payload, "decisionKind");
|
|
1271
|
+
if (!isBeliefDecisionKind(decisionKind)) {
|
|
1272
|
+
return {
|
|
1273
|
+
diagnostic: {
|
|
1274
|
+
runId: event.runId,
|
|
1275
|
+
eventId: event.eventId,
|
|
1276
|
+
severity: "warning",
|
|
1277
|
+
reason: `belief decision event has unsupported decisionKind "${decisionKind ?? ""}"`
|
|
1278
|
+
}
|
|
1279
|
+
};
|
|
1280
|
+
}
|
|
1281
|
+
const chosenAction = stringField2(payload, "chosenAction");
|
|
1282
|
+
if (!chosenAction) {
|
|
1283
|
+
return {
|
|
1284
|
+
diagnostic: {
|
|
1285
|
+
runId: event.runId,
|
|
1286
|
+
eventId: event.eventId,
|
|
1287
|
+
severity: "warning",
|
|
1288
|
+
reason: "belief decision event is missing chosenAction"
|
|
1289
|
+
}
|
|
1290
|
+
};
|
|
1291
|
+
}
|
|
1292
|
+
const evidence = [
|
|
1293
|
+
{
|
|
1294
|
+
source: "event",
|
|
1295
|
+
id: event.eventId,
|
|
1296
|
+
runId: event.runId,
|
|
1297
|
+
eventId: event.eventId,
|
|
1298
|
+
quality: "direct"
|
|
1299
|
+
}
|
|
1300
|
+
];
|
|
1301
|
+
if (event.spanId && context.spanExists) {
|
|
1302
|
+
evidence.push({
|
|
1303
|
+
source: "span",
|
|
1304
|
+
id: event.spanId,
|
|
1305
|
+
runId: event.runId,
|
|
1306
|
+
spanId: event.spanId,
|
|
1307
|
+
quality: "direct"
|
|
1308
|
+
});
|
|
1309
|
+
}
|
|
1310
|
+
return {
|
|
1311
|
+
decision: {
|
|
1312
|
+
id: stringField2(payload, "id") ?? event.eventId,
|
|
1313
|
+
runId: event.runId,
|
|
1314
|
+
scenarioId: stringField2(payload, "scenarioId") ?? context.scenarioId,
|
|
1315
|
+
stepIndex: numberField2(payload, "stepIndex") ?? context.stepIndex,
|
|
1316
|
+
kind: decisionKind,
|
|
1317
|
+
chosenAction,
|
|
1318
|
+
candidateActions: stringArrayField(payload, "candidateActions"),
|
|
1319
|
+
confidence: finiteUnitField(payload, "confidence"),
|
|
1320
|
+
behaviorProb: numberField2(payload, "behaviorProb"),
|
|
1321
|
+
targetProb: numberField2(payload, "targetProb"),
|
|
1322
|
+
qHat: finiteUnitField(payload, "qHat"),
|
|
1323
|
+
costUsd: nonNegativeNumberField(payload, "costUsd"),
|
|
1324
|
+
evidence,
|
|
1325
|
+
outcome: parseOutcome(payload),
|
|
1326
|
+
metadata: recordField(payload, "metadata")
|
|
1327
|
+
}
|
|
1328
|
+
};
|
|
1329
|
+
}
|
|
1330
|
+
function parseOutcome(payload) {
|
|
1331
|
+
const value = recordField(payload, "outcome");
|
|
1332
|
+
if (!value) return void 0;
|
|
1333
|
+
return {
|
|
1334
|
+
success: typeof value.success === "boolean" ? value.success : void 0,
|
|
1335
|
+
score: finiteUnitField(value, "score"),
|
|
1336
|
+
reward: finiteUnitField(value, "reward"),
|
|
1337
|
+
costUsd: nonNegativeNumberField(value, "costUsd"),
|
|
1338
|
+
observedAt: stringField2(value, "observedAt"),
|
|
1339
|
+
metadata: recordField(value, "metadata")
|
|
1340
|
+
};
|
|
1341
|
+
}
|
|
1342
|
+
function stringField2(obj, key) {
|
|
1343
|
+
const value = obj[key];
|
|
1344
|
+
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
1345
|
+
}
|
|
1346
|
+
function numberField2(obj, key) {
|
|
1347
|
+
const value = obj[key];
|
|
1348
|
+
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
1349
|
+
}
|
|
1350
|
+
function finiteUnitField(obj, key) {
|
|
1351
|
+
const value = numberField2(obj, key);
|
|
1352
|
+
return value === void 0 ? void 0 : Math.max(0, Math.min(1, value));
|
|
1353
|
+
}
|
|
1354
|
+
function nonNegativeNumberField(obj, key) {
|
|
1355
|
+
const value = numberField2(obj, key);
|
|
1356
|
+
return value === void 0 ? void 0 : Math.max(0, value);
|
|
1357
|
+
}
|
|
1358
|
+
function stringArrayField(obj, key) {
|
|
1359
|
+
const value = obj[key];
|
|
1360
|
+
if (!Array.isArray(value)) return void 0;
|
|
1361
|
+
const strings = value.filter(
|
|
1362
|
+
(item) => typeof item === "string" && item.length > 0
|
|
1363
|
+
);
|
|
1364
|
+
return strings.length > 0 ? strings : void 0;
|
|
1365
|
+
}
|
|
1366
|
+
function recordField(obj, key) {
|
|
1367
|
+
const value = obj[key];
|
|
1368
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) return void 0;
|
|
1369
|
+
return value;
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
// src/belief-state/runtime-hooks.ts
|
|
1373
|
+
var DEFAULT_MAX_CONTEXT_CHARS = 12e3;
|
|
1374
|
+
var DEFAULT_PAYLOAD_PREVIEW_CHARS = 2e3;
|
|
1375
|
+
function runtimeDecisionPointToBeliefShadowProbeInput(point, options) {
|
|
1376
|
+
const diagnostics = [];
|
|
1377
|
+
const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
|
|
1378
|
+
if (!decisionKind) return { diagnostics };
|
|
1379
|
+
const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
|
|
1380
|
+
const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
|
|
1381
|
+
return {
|
|
1382
|
+
input: {
|
|
1383
|
+
probeId: options.probeId,
|
|
1384
|
+
decisionId: point.id,
|
|
1385
|
+
runId: point.runId,
|
|
1386
|
+
scenarioId: point.scenarioId,
|
|
1387
|
+
stepIndex: point.stepIndex,
|
|
1388
|
+
decisionKind,
|
|
1389
|
+
candidateActions: uniqueStrings(point.candidateActions ?? []),
|
|
1390
|
+
evidence: evidence.map((ref) => ({
|
|
1391
|
+
id: ref.id,
|
|
1392
|
+
source: ref.source,
|
|
1393
|
+
...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
|
|
1394
|
+
...ref.quality ? { quality: ref.quality } : {}
|
|
1395
|
+
})),
|
|
1396
|
+
context: trimText(point.context, options.maxContextChars),
|
|
1397
|
+
metadata: mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence))
|
|
1398
|
+
},
|
|
1399
|
+
diagnostics
|
|
1400
|
+
};
|
|
1401
|
+
}
|
|
1402
|
+
function runtimeDecisionPointToBeliefDecisionPoint(point, options) {
|
|
1403
|
+
const diagnostics = [];
|
|
1404
|
+
const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
|
|
1405
|
+
const chosenAction = stringOrUndefined(options.chosenAction);
|
|
1406
|
+
if (!chosenAction) {
|
|
1407
|
+
diagnostics.push({
|
|
1408
|
+
decisionId: point.id,
|
|
1409
|
+
severity: "error",
|
|
1410
|
+
reason: "missing chosenAction"
|
|
1411
|
+
});
|
|
1412
|
+
}
|
|
1413
|
+
const candidateActions = uniqueStrings(point.candidateActions ?? []);
|
|
1414
|
+
if (chosenAction && candidateActions.length > 0 && !candidateActions.includes(chosenAction)) {
|
|
1415
|
+
diagnostics.push({
|
|
1416
|
+
decisionId: point.id,
|
|
1417
|
+
severity: "warning",
|
|
1418
|
+
reason: `chosenAction ${chosenAction} is not in candidateActions`
|
|
1419
|
+
});
|
|
1420
|
+
}
|
|
1421
|
+
if (!decisionKind || !chosenAction) return { diagnostics };
|
|
1422
|
+
const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
|
|
1423
|
+
const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
|
|
1424
|
+
return {
|
|
1425
|
+
point: {
|
|
1426
|
+
id: point.id,
|
|
1427
|
+
runId: point.runId,
|
|
1428
|
+
scenarioId: point.scenarioId,
|
|
1429
|
+
stepIndex: point.stepIndex,
|
|
1430
|
+
kind: decisionKind,
|
|
1431
|
+
chosenAction,
|
|
1432
|
+
candidateActions,
|
|
1433
|
+
confidence: unitProbabilityOrUndefined(options.confidence),
|
|
1434
|
+
behaviorProb: finiteNumberOrUndefined(options.behaviorProb),
|
|
1435
|
+
targetProb: finiteNumberOrUndefined(options.targetProb),
|
|
1436
|
+
qHat: options.qHat === null ? null : unitProbabilityOrUndefined(options.qHat),
|
|
1437
|
+
costUsd: nonNegativeNumberOrUndefined(options.costUsd),
|
|
1438
|
+
evidence: evidence.map((ref) => runtimeEvidenceToBeliefEvidence(ref, point)),
|
|
1439
|
+
outcome: options.outcome,
|
|
1440
|
+
metadata: mergeMetadata(
|
|
1441
|
+
mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence)),
|
|
1442
|
+
options.metadata
|
|
1443
|
+
)
|
|
1444
|
+
},
|
|
1445
|
+
diagnostics
|
|
1446
|
+
};
|
|
1447
|
+
}
|
|
1448
|
+
function createBeliefRuntimeHookCollector(defaults) {
|
|
1449
|
+
const decisions = [];
|
|
1450
|
+
const events = [];
|
|
1451
|
+
return {
|
|
1452
|
+
hooks: {
|
|
1453
|
+
onEvent: (event) => {
|
|
1454
|
+
events.push(snapshotRuntimeHookEvent(event));
|
|
1455
|
+
},
|
|
1456
|
+
onDecisionPoint: (point) => {
|
|
1457
|
+
decisions.push(snapshotRuntimeDecisionPoint(point));
|
|
1458
|
+
}
|
|
1459
|
+
},
|
|
1460
|
+
decisions,
|
|
1461
|
+
events,
|
|
1462
|
+
toShadowProbeInputs: (options = {}) => {
|
|
1463
|
+
const inputs = [];
|
|
1464
|
+
const diagnostics = [];
|
|
1465
|
+
const includeLifecycleEvidence = options.includeLifecycleEvidence ?? defaults.includeLifecycleEvidence;
|
|
1466
|
+
for (const point of decisions) {
|
|
1467
|
+
const report = runtimeDecisionPointToBeliefShadowProbeInput(point, {
|
|
1468
|
+
...defaults,
|
|
1469
|
+
...options,
|
|
1470
|
+
includeLifecycleEvidence,
|
|
1471
|
+
lifecycleEvents: includeLifecycleEvidence === false ? void 0 : options.lifecycleEvents ?? defaults.lifecycleEvents ?? events
|
|
1472
|
+
});
|
|
1473
|
+
if (report.input) inputs.push(report.input);
|
|
1474
|
+
diagnostics.push(...report.diagnostics);
|
|
1475
|
+
}
|
|
1476
|
+
return { inputs, diagnostics };
|
|
1477
|
+
},
|
|
1478
|
+
clear: () => {
|
|
1479
|
+
decisions.length = 0;
|
|
1480
|
+
events.length = 0;
|
|
1481
|
+
}
|
|
1482
|
+
};
|
|
1483
|
+
}
|
|
1484
|
+
function resolveDecisionKind(point, override, diagnostics) {
|
|
1485
|
+
const kind = override ?? point.kind;
|
|
1486
|
+
if (isBeliefDecisionKind(kind)) return kind;
|
|
1487
|
+
diagnostics.push({
|
|
1488
|
+
decisionId: point.id,
|
|
1489
|
+
severity: "error",
|
|
1490
|
+
reason: `unsupported decisionKind "${kind}"`
|
|
1491
|
+
});
|
|
1492
|
+
return void 0;
|
|
1493
|
+
}
|
|
1494
|
+
function runtimeEvidenceToBeliefEvidence(ref, point) {
|
|
1495
|
+
if (isBeliefEvidenceSource(ref.source)) {
|
|
1496
|
+
return {
|
|
1497
|
+
source: ref.source,
|
|
1498
|
+
id: ref.id,
|
|
1499
|
+
runId: point.runId,
|
|
1500
|
+
detail: ref.detail,
|
|
1501
|
+
quality: ref.quality,
|
|
1502
|
+
metadata: ref.metadata
|
|
1503
|
+
};
|
|
1504
|
+
}
|
|
1505
|
+
return {
|
|
1506
|
+
source: "event",
|
|
1507
|
+
id: ref.id,
|
|
1508
|
+
runId: point.runId,
|
|
1509
|
+
detail: ref.detail,
|
|
1510
|
+
quality: ref.quality,
|
|
1511
|
+
metadata: mergeMetadata({ runtimeSource: ref.source }, ref.metadata)
|
|
1512
|
+
};
|
|
1513
|
+
}
|
|
1514
|
+
function runtimeHookEventsToEvidenceRefs(point, options) {
|
|
1515
|
+
if (options.includeLifecycleEvidence === false) return [];
|
|
1516
|
+
return (options.lifecycleEvents ?? []).filter((event) => runtimeHookEventMatchesDecision(point, event)).map(runtimeHookEventToEvidenceRef);
|
|
1517
|
+
}
|
|
1518
|
+
function runtimeHookEventMatchesDecision(point, event) {
|
|
1519
|
+
if (event.runId !== point.runId) return false;
|
|
1520
|
+
if (event.scenarioId && point.scenarioId && event.scenarioId !== point.scenarioId) return false;
|
|
1521
|
+
return event.stepIndex === void 0 || event.stepIndex === point.stepIndex;
|
|
1522
|
+
}
|
|
1523
|
+
function runtimeHookEventToEvidenceRef(event) {
|
|
1524
|
+
return {
|
|
1525
|
+
source: "runtime_event",
|
|
1526
|
+
id: event.id,
|
|
1527
|
+
detail: `${event.target}:${event.phase}`,
|
|
1528
|
+
quality: "direct",
|
|
1529
|
+
metadata: mergeMetadata(
|
|
1530
|
+
compactMetadata({
|
|
1531
|
+
target: event.target,
|
|
1532
|
+
phase: event.phase,
|
|
1533
|
+
timestamp: event.timestamp,
|
|
1534
|
+
stepIndex: event.stepIndex,
|
|
1535
|
+
parentId: event.parentId,
|
|
1536
|
+
payloadPreview: previewUnknown(event.payload)
|
|
1537
|
+
}),
|
|
1538
|
+
event.metadata
|
|
1539
|
+
)
|
|
1540
|
+
};
|
|
1541
|
+
}
|
|
1542
|
+
function lifecycleMetadata(refs) {
|
|
1543
|
+
if (refs.length === 0) return void 0;
|
|
1544
|
+
return {
|
|
1545
|
+
lifecycleEventCount: refs.length,
|
|
1546
|
+
lifecycleEventIds: refs.map((ref) => ref.id)
|
|
1547
|
+
};
|
|
1548
|
+
}
|
|
1549
|
+
function snapshotRuntimeHookEvent(event) {
|
|
1550
|
+
return {
|
|
1551
|
+
id: event.id,
|
|
1552
|
+
runId: event.runId,
|
|
1553
|
+
scenarioId: event.scenarioId,
|
|
1554
|
+
target: event.target,
|
|
1555
|
+
phase: event.phase,
|
|
1556
|
+
timestamp: event.timestamp,
|
|
1557
|
+
stepIndex: event.stepIndex,
|
|
1558
|
+
parentId: event.parentId,
|
|
1559
|
+
payload: snapshotUnknown(event.payload),
|
|
1560
|
+
metadata: event.metadata ? { ...event.metadata } : void 0
|
|
1561
|
+
};
|
|
1562
|
+
}
|
|
1563
|
+
function snapshotRuntimeDecisionPoint(point) {
|
|
1564
|
+
return {
|
|
1565
|
+
id: point.id,
|
|
1566
|
+
runId: point.runId,
|
|
1567
|
+
scenarioId: point.scenarioId,
|
|
1568
|
+
stepIndex: point.stepIndex,
|
|
1569
|
+
kind: point.kind,
|
|
1570
|
+
candidateActions: [...point.candidateActions ?? []],
|
|
1571
|
+
context: point.context,
|
|
1572
|
+
evidence: (point.evidence ?? []).map((ref) => ({
|
|
1573
|
+
source: ref.source,
|
|
1574
|
+
id: ref.id,
|
|
1575
|
+
detail: ref.detail,
|
|
1576
|
+
quality: ref.quality,
|
|
1577
|
+
metadata: ref.metadata ? { ...ref.metadata } : void 0
|
|
1578
|
+
})),
|
|
1579
|
+
metadata: point.metadata ? { ...point.metadata } : void 0
|
|
1580
|
+
};
|
|
1581
|
+
}
|
|
1582
|
+
function mergeMetadata(base, extra) {
|
|
1583
|
+
if (!base && !extra) return void 0;
|
|
1584
|
+
return { ...base ?? {}, ...extra ?? {} };
|
|
1585
|
+
}
|
|
1586
|
+
function compactMetadata(values) {
|
|
1587
|
+
const entries = Object.entries(values).filter(([, value]) => value !== void 0);
|
|
1588
|
+
return entries.length > 0 ? Object.fromEntries(entries) : void 0;
|
|
1589
|
+
}
|
|
1590
|
+
function previewUnknown(value, maxChars = DEFAULT_PAYLOAD_PREVIEW_CHARS) {
|
|
1591
|
+
if (value === void 0) return void 0;
|
|
1592
|
+
if (typeof value === "string") return trimText(value, maxChars);
|
|
1593
|
+
try {
|
|
1594
|
+
return trimText(JSON.stringify(value), maxChars);
|
|
1595
|
+
} catch {
|
|
1596
|
+
return trimText(String(value), maxChars);
|
|
1597
|
+
}
|
|
1598
|
+
}
|
|
1599
|
+
function snapshotUnknown(value) {
|
|
1600
|
+
if (Array.isArray(value)) return [...value];
|
|
1601
|
+
if (isRecord2(value)) return { ...value };
|
|
1602
|
+
return value;
|
|
1603
|
+
}
|
|
1604
|
+
function isRecord2(value) {
|
|
1605
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1606
|
+
}
|
|
1607
|
+
function uniqueStrings(values) {
|
|
1608
|
+
return [...new Set(values.filter((value) => value.length > 0))];
|
|
1609
|
+
}
|
|
1610
|
+
function trimText(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS) {
|
|
1611
|
+
if (!value) return void 0;
|
|
1612
|
+
return value.length > maxChars ? value.slice(value.length - maxChars) : value;
|
|
1613
|
+
}
|
|
1614
|
+
function stringOrUndefined(value) {
|
|
1615
|
+
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
1616
|
+
}
|
|
1617
|
+
function finiteNumberOrUndefined(value) {
|
|
1618
|
+
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
1619
|
+
}
|
|
1620
|
+
function unitProbabilityOrUndefined(value) {
|
|
1621
|
+
const number = finiteNumberOrUndefined(value);
|
|
1622
|
+
return number !== void 0 && number >= 0 && number <= 1 ? number : void 0;
|
|
1623
|
+
}
|
|
1624
|
+
function nonNegativeNumberOrUndefined(value) {
|
|
1625
|
+
const number = finiteNumberOrUndefined(value);
|
|
1626
|
+
return number !== void 0 && number >= 0 ? number : void 0;
|
|
1627
|
+
}
|
|
1628
|
+
|
|
1629
|
+
// src/belief-state/phase0-measurement.ts
|
|
1630
|
+
var DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
|
|
1631
|
+
function buildRuntimeBeliefPhase0Measurement(options) {
|
|
1632
|
+
const runsById = new Map(options.runs.map((run) => [run.runId, run]));
|
|
1633
|
+
const labelsByDecisionId = /* @__PURE__ */ new Map();
|
|
1634
|
+
const diagnostics = [];
|
|
1635
|
+
for (const label of options.labels) {
|
|
1636
|
+
if (labelsByDecisionId.has(label.decisionId)) {
|
|
1637
|
+
diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
|
|
1638
|
+
}
|
|
1639
|
+
labelsByDecisionId.set(label.decisionId, label);
|
|
1640
|
+
}
|
|
1641
|
+
const points = [];
|
|
1642
|
+
let missingRunRecordCount = 0;
|
|
1643
|
+
let missingLabelCount = 0;
|
|
1644
|
+
for (const decision of options.decisions) {
|
|
1645
|
+
const run = runsById.get(decision.runId);
|
|
1646
|
+
if (!run) {
|
|
1647
|
+
missingRunRecordCount += 1;
|
|
1648
|
+
diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
|
|
1649
|
+
continue;
|
|
1650
|
+
}
|
|
1651
|
+
const label = labelsByDecisionId.get(decision.id);
|
|
1652
|
+
if (!label) {
|
|
1653
|
+
missingLabelCount += 1;
|
|
1654
|
+
diagnostics.push(`${decision.id}: missing observed action/outcome label`);
|
|
1655
|
+
continue;
|
|
1656
|
+
}
|
|
1657
|
+
const splitTag = label.splitTag ?? run.splitTag;
|
|
1658
|
+
const report = runtimeDecisionPointToBeliefDecisionPoint(
|
|
1659
|
+
{ ...decision, scenarioId: decision.scenarioId ?? run.scenarioId },
|
|
1660
|
+
{
|
|
1661
|
+
chosenAction: label.chosenAction,
|
|
1662
|
+
confidence: label.confidence,
|
|
1663
|
+
behaviorProb: label.behaviorProb,
|
|
1664
|
+
targetProb: label.targetProb,
|
|
1665
|
+
qHat: label.qHat,
|
|
1666
|
+
costUsd: label.costUsd,
|
|
1667
|
+
outcome: label.outcome,
|
|
1668
|
+
lifecycleEvents: options.events,
|
|
1669
|
+
metadata: compactMetadata2({
|
|
1670
|
+
baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
|
|
1671
|
+
splitTag,
|
|
1672
|
+
...label.metadata
|
|
1673
|
+
})
|
|
1674
|
+
}
|
|
1675
|
+
);
|
|
1676
|
+
diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
|
|
1677
|
+
if (report.point) points.push(report.point);
|
|
1678
|
+
}
|
|
1679
|
+
const packet = buildBeliefDecisionResearchEvidencePacket({
|
|
1680
|
+
...options,
|
|
1681
|
+
points
|
|
1682
|
+
});
|
|
1683
|
+
return {
|
|
1684
|
+
points,
|
|
1685
|
+
packet,
|
|
1686
|
+
summary: summarizePhase0Measurement(options, points, packet, {
|
|
1687
|
+
missingRunRecordCount,
|
|
1688
|
+
missingLabelCount
|
|
1689
|
+
}),
|
|
1690
|
+
diagnostics
|
|
1691
|
+
};
|
|
1692
|
+
}
|
|
1693
|
+
function summarizePhase0Measurement(options, points, packet, counts) {
|
|
1694
|
+
const producerDecisionCount = options.decisions.length;
|
|
1695
|
+
return {
|
|
1696
|
+
runCount: options.runs.length,
|
|
1697
|
+
producerDecisionCount,
|
|
1698
|
+
lifecycleEventCount: options.events?.length ?? 0,
|
|
1699
|
+
labelCount: options.labels.length,
|
|
1700
|
+
completedPointCount: points.length,
|
|
1701
|
+
runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
|
|
1702
|
+
labelJoinRate: ratio(points.length, producerDecisionCount),
|
|
1703
|
+
missingRunRecordCount: counts.missingRunRecordCount,
|
|
1704
|
+
missingLabelCount: counts.missingLabelCount,
|
|
1705
|
+
withEvidence: points.filter((point) => point.evidence.length > 0).length,
|
|
1706
|
+
withOutcome: points.filter((point) => point.outcome).length,
|
|
1707
|
+
withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
|
|
1708
|
+
withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
|
|
1709
|
+
withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
|
|
1710
|
+
baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
|
|
1711
|
+
packetStatus: packet.status,
|
|
1712
|
+
claimScope: packet.claimScope
|
|
1713
|
+
};
|
|
1714
|
+
}
|
|
1715
|
+
function ratio(numerator, denominator) {
|
|
1716
|
+
return denominator > 0 ? numerator / denominator : 0;
|
|
1717
|
+
}
|
|
1718
|
+
function compactMetadata2(values) {
|
|
1719
|
+
const entries = Object.entries(values).filter(([, value]) => value !== void 0);
|
|
1720
|
+
return entries.length > 0 ? Object.fromEntries(entries) : void 0;
|
|
1721
|
+
}
|
|
1722
|
+
|
|
1723
|
+
// src/belief-state/runtime-benchmark-corpus.ts
|
|
1724
|
+
var MAX_STRING_LENGTH = 12e3;
|
|
1725
|
+
var MAX_CONTEXT_LENGTH = 2e4;
|
|
1726
|
+
var MAX_EVIDENCE_DETAIL_LENGTH = 2e3;
|
|
1727
|
+
var MAX_CANDIDATE_ACTIONS = 50;
|
|
1728
|
+
var MAX_EVIDENCE_REFS = 50;
|
|
1729
|
+
var MAX_METADATA_DEPTH = 4;
|
|
1730
|
+
var MAX_METADATA_KEYS = 100;
|
|
1731
|
+
var SENSITIVE_KEY_RE = /(?:authorization|api[_-]?key|token|secret|password|cookie|credential|bearer)/i;
|
|
1732
|
+
var SENSITIVE_VALUE_RES = [
|
|
1733
|
+
/\bBearer\s+[A-Za-z0-9._~+/=-]+/gi,
|
|
1734
|
+
/\b(?:sk|gh[pousr])_[A-Za-z0-9_]{20,}\b/g,
|
|
1735
|
+
/\b(?:sk|ghp|gho|ghu|ghs|ghr)-[A-Za-z0-9_-]{20,}\b/g
|
|
1736
|
+
];
|
|
1737
|
+
var SENSITIVE_ASSIGNMENT_RE = /\b(api[_-]?key|token|secret|password|cookie)\s*[:=]\s*["']?[^"'\s,;}]+/gi;
|
|
1738
|
+
function buildRuntimeBenchmarkBeliefPhase0Measurement(options) {
|
|
1739
|
+
const diagnostics = [];
|
|
1740
|
+
const trajectory = projectRuntimeTrajectoryEvidence({
|
|
1741
|
+
records: options.records,
|
|
1742
|
+
defaultSplitTag: options.defaultSplitTag,
|
|
1743
|
+
recordIdOf: runtimeBenchmarkRecordId,
|
|
1744
|
+
scenarioIdOf: runtimeBenchmarkScenarioId
|
|
1745
|
+
});
|
|
1746
|
+
const decisions = options.decisions ?? runtimeBenchmarkDecisionPoints(options.records, diagnostics);
|
|
1747
|
+
const labels = options.labels ?? [];
|
|
1748
|
+
if (decisions.length === 0) {
|
|
1749
|
+
diagnostics.push(
|
|
1750
|
+
"no runtime decision points supplied or found on records; benchmark lifecycle events alone cannot produce belief decision rows"
|
|
1751
|
+
);
|
|
1752
|
+
}
|
|
1753
|
+
if (labels.length === 0 && decisions.length > 0) {
|
|
1754
|
+
diagnostics.push(
|
|
1755
|
+
"no decision labels supplied; observed action/outcome joins will be incomplete"
|
|
1756
|
+
);
|
|
1757
|
+
}
|
|
1758
|
+
const measurement = buildRuntimeBeliefPhase0Measurement({
|
|
1759
|
+
...options,
|
|
1760
|
+
runs: trajectory.runs,
|
|
1761
|
+
events: trajectory.events,
|
|
1762
|
+
decisions,
|
|
1763
|
+
labels
|
|
1764
|
+
});
|
|
1765
|
+
return {
|
|
1766
|
+
runs: trajectory.runs,
|
|
1767
|
+
events: trajectory.events,
|
|
1768
|
+
decisions,
|
|
1769
|
+
labels,
|
|
1770
|
+
trajectory,
|
|
1771
|
+
measurement,
|
|
1772
|
+
summary: {
|
|
1773
|
+
decisionCount: decisions.length,
|
|
1774
|
+
labelCount: labels.length
|
|
1775
|
+
},
|
|
1776
|
+
diagnostics: [...trajectory.diagnostics, ...diagnostics, ...measurement.diagnostics]
|
|
1777
|
+
};
|
|
1778
|
+
}
|
|
1779
|
+
function runtimeBenchmarkRecordId(record2) {
|
|
1780
|
+
const parts = [
|
|
1781
|
+
nonEmptyString(record2.benchmark),
|
|
1782
|
+
nonEmptyString(record2.instanceId),
|
|
1783
|
+
nonEmptyString(record2.condition)
|
|
1784
|
+
].filter((part) => part !== void 0);
|
|
1785
|
+
return parts.length > 0 ? parts.join(":") : void 0;
|
|
1786
|
+
}
|
|
1787
|
+
function runtimeBenchmarkScenarioId(record2) {
|
|
1788
|
+
return nonEmptyString(record2.instanceId);
|
|
1789
|
+
}
|
|
1790
|
+
function runtimeBenchmarkDecisionPoints(records, diagnostics) {
|
|
1791
|
+
const decisions = [];
|
|
1792
|
+
for (let recordIndex = 0; recordIndex < records.length; recordIndex += 1) {
|
|
1793
|
+
const record2 = records[recordIndex];
|
|
1794
|
+
const raw = record2.runtimeDecisionPoints;
|
|
1795
|
+
if (raw === void 0) continue;
|
|
1796
|
+
const recordId = runtimeBenchmarkRecordId(record2) ?? `record[${recordIndex}]`;
|
|
1797
|
+
if (!Array.isArray(raw)) {
|
|
1798
|
+
diagnostics.push(`${recordId}: runtimeDecisionPoints is not an array`);
|
|
1799
|
+
continue;
|
|
1800
|
+
}
|
|
1801
|
+
for (let pointIndex = 0; pointIndex < raw.length; pointIndex += 1) {
|
|
1802
|
+
const point = runtimeBenchmarkDecisionPoint(raw[pointIndex], {
|
|
1803
|
+
diagnostics,
|
|
1804
|
+
path: `${recordId}: runtimeDecisionPoints[${pointIndex}]`
|
|
1805
|
+
});
|
|
1806
|
+
if (!point) {
|
|
1807
|
+
diagnostics.push(
|
|
1808
|
+
`${recordId}: runtimeDecisionPoints[${pointIndex}] is not a RuntimeDecisionPoint`
|
|
1809
|
+
);
|
|
1810
|
+
continue;
|
|
1811
|
+
}
|
|
1812
|
+
decisions.push(point);
|
|
1813
|
+
}
|
|
1814
|
+
}
|
|
1815
|
+
return decisions;
|
|
1816
|
+
}
|
|
1817
|
+
function runtimeBenchmarkDecisionPoint(input, context) {
|
|
1818
|
+
if (!isRecord3(input)) return null;
|
|
1819
|
+
if (typeof input.id !== "string" || input.id.length === 0) return null;
|
|
1820
|
+
if (typeof input.runId !== "string" || input.runId.length === 0) return null;
|
|
1821
|
+
if (typeof input.stepIndex !== "number" || !Number.isInteger(input.stepIndex) || input.stepIndex < 0) {
|
|
1822
|
+
return null;
|
|
1823
|
+
}
|
|
1824
|
+
if (typeof input.kind !== "string" || input.kind.length === 0) return null;
|
|
1825
|
+
return {
|
|
1826
|
+
id: sanitizeString(input.id, MAX_STRING_LENGTH),
|
|
1827
|
+
runId: sanitizeString(input.runId, MAX_STRING_LENGTH),
|
|
1828
|
+
scenarioId: sanitizeOptionalString(input.scenarioId, MAX_STRING_LENGTH),
|
|
1829
|
+
stepIndex: input.stepIndex,
|
|
1830
|
+
kind: sanitizeString(input.kind, MAX_STRING_LENGTH),
|
|
1831
|
+
candidateActions: stringArray(input.candidateActions, {
|
|
1832
|
+
...context,
|
|
1833
|
+
maxItems: MAX_CANDIDATE_ACTIONS,
|
|
1834
|
+
label: "candidateActions"
|
|
1835
|
+
}),
|
|
1836
|
+
context: sanitizeOptionalString(input.context, MAX_CONTEXT_LENGTH),
|
|
1837
|
+
evidence: runtimeBenchmarkEvidence(input.evidence, context),
|
|
1838
|
+
metadata: sanitizeMetadataRecord(input.metadata)
|
|
1839
|
+
};
|
|
1840
|
+
}
|
|
1841
|
+
function runtimeBenchmarkEvidence(input, context) {
|
|
1842
|
+
if (!Array.isArray(input)) return [];
|
|
1843
|
+
if (input.length > MAX_EVIDENCE_REFS) {
|
|
1844
|
+
context.diagnostics.push(`${context.path}: evidence truncated to ${MAX_EVIDENCE_REFS} refs`);
|
|
1845
|
+
}
|
|
1846
|
+
return input.slice(0, MAX_EVIDENCE_REFS).flatMap((item) => {
|
|
1847
|
+
if (!isRecord3(item)) return [];
|
|
1848
|
+
const source = sanitizeOptionalString(item.source, MAX_STRING_LENGTH);
|
|
1849
|
+
const id = sanitizeOptionalString(item.id, MAX_STRING_LENGTH);
|
|
1850
|
+
if (!source || !id) return [];
|
|
1851
|
+
return [
|
|
1852
|
+
{
|
|
1853
|
+
source,
|
|
1854
|
+
id,
|
|
1855
|
+
detail: sanitizeOptionalString(item.detail, MAX_EVIDENCE_DETAIL_LENGTH),
|
|
1856
|
+
metadata: sanitizeMetadataRecord(item.metadata)
|
|
1857
|
+
}
|
|
1858
|
+
];
|
|
1859
|
+
});
|
|
1860
|
+
}
|
|
1861
|
+
function stringArray(input, context) {
|
|
1862
|
+
if (!Array.isArray(input)) return void 0;
|
|
1863
|
+
if (input.length > context.maxItems) {
|
|
1864
|
+
context.diagnostics.push(`${context.path}: ${context.label} truncated to ${context.maxItems}`);
|
|
1865
|
+
}
|
|
1866
|
+
const values = input.slice(0, context.maxItems).filter((value) => typeof value === "string" && value.length > 0).map((value) => sanitizeString(value, MAX_STRING_LENGTH));
|
|
1867
|
+
return values.length > 0 ? values : void 0;
|
|
1868
|
+
}
|
|
1869
|
+
function sanitizeMetadataRecord(metadata) {
|
|
1870
|
+
if (!isRecord3(metadata)) return void 0;
|
|
1871
|
+
const sanitized = sanitizeMetadata(metadata);
|
|
1872
|
+
if (!sanitized || typeof sanitized !== "object" || Array.isArray(sanitized)) return void 0;
|
|
1873
|
+
return sanitized;
|
|
1874
|
+
}
|
|
1875
|
+
function sanitizeMetadata(value, depth = 0) {
|
|
1876
|
+
if (value == null) return value;
|
|
1877
|
+
if (typeof value === "string") return sanitizeString(value, MAX_STRING_LENGTH);
|
|
1878
|
+
if (typeof value === "number" || typeof value === "boolean") return value;
|
|
1879
|
+
if (Array.isArray(value)) {
|
|
1880
|
+
if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
|
|
1881
|
+
return value.slice(0, MAX_METADATA_KEYS).map((item) => sanitizeMetadata(item, depth + 1));
|
|
1882
|
+
}
|
|
1883
|
+
if (!isRecord3(value)) return void 0;
|
|
1884
|
+
if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
|
|
1885
|
+
const sanitized = {};
|
|
1886
|
+
for (const [key, nested] of Object.entries(value).slice(0, MAX_METADATA_KEYS)) {
|
|
1887
|
+
sanitized[key] = SENSITIVE_KEY_RE.test(key) ? "[REDACTED]" : sanitizeMetadata(nested, depth + 1);
|
|
1888
|
+
}
|
|
1889
|
+
return sanitized;
|
|
1890
|
+
}
|
|
1891
|
+
function sanitizeOptionalString(value, maxLength) {
|
|
1892
|
+
return typeof value === "string" && value.length > 0 ? sanitizeString(value, maxLength) : void 0;
|
|
1893
|
+
}
|
|
1894
|
+
function sanitizeString(value, maxLength) {
|
|
1895
|
+
let sanitized = value;
|
|
1896
|
+
for (const pattern of SENSITIVE_VALUE_RES) {
|
|
1897
|
+
sanitized = sanitized.replace(pattern, "[REDACTED]");
|
|
1898
|
+
}
|
|
1899
|
+
sanitized = sanitized.replace(
|
|
1900
|
+
SENSITIVE_ASSIGNMENT_RE,
|
|
1901
|
+
(_match, key) => `${key}=[REDACTED]`
|
|
1902
|
+
);
|
|
1903
|
+
if (sanitized.length <= maxLength) return sanitized;
|
|
1904
|
+
return sanitized.slice(0, maxLength);
|
|
1905
|
+
}
|
|
1906
|
+
function isRecord3(value) {
|
|
1907
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1908
|
+
}
|
|
1909
|
+
function nonEmptyString(value) {
|
|
1910
|
+
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
1911
|
+
}
|
|
1912
|
+
|
|
1913
|
+
// src/belief-state/shadow-probe.ts
|
|
1914
|
+
var DEFAULT_CONCURRENCY = 4;
|
|
1915
|
+
var DEFAULT_MAX_CONTEXT_CHARS2 = 12e3;
|
|
1916
|
+
async function runBeliefShadowProbe(options) {
|
|
1917
|
+
const concurrency = boundedInteger(options.concurrency ?? DEFAULT_CONCURRENCY, 1, 32);
|
|
1918
|
+
const records = [];
|
|
1919
|
+
const diagnostics = [];
|
|
1920
|
+
let next = 0;
|
|
1921
|
+
async function worker() {
|
|
1922
|
+
while (next < options.points.length) {
|
|
1923
|
+
const index = next;
|
|
1924
|
+
next += 1;
|
|
1925
|
+
const point = options.points[index];
|
|
1926
|
+
if (!point) continue;
|
|
1927
|
+
const result = await probePoint(point, options);
|
|
1928
|
+
records[index] = result.record;
|
|
1929
|
+
diagnostics.push(...result.diagnostics);
|
|
1930
|
+
}
|
|
1931
|
+
}
|
|
1932
|
+
await Promise.all(Array.from({ length: Math.min(concurrency, options.points.length) }, worker));
|
|
1933
|
+
const completed = records.filter((record2) => !!record2);
|
|
1934
|
+
return {
|
|
1935
|
+
probeId: options.probeId,
|
|
1936
|
+
records: completed,
|
|
1937
|
+
diagnostics,
|
|
1938
|
+
summary: summarizeShadowProbe(options.points.length, completed)
|
|
1939
|
+
};
|
|
1940
|
+
}
|
|
1941
|
+
function formatBeliefShadowProbePrompt(input) {
|
|
1942
|
+
return [
|
|
1943
|
+
"Return only JSON. Do not include chain-of-thought.",
|
|
1944
|
+
"Infer the agent belief state at this decision boundary using only the context below.",
|
|
1945
|
+
"",
|
|
1946
|
+
`decisionKind: ${input.decisionKind}`,
|
|
1947
|
+
`candidateActions: ${JSON.stringify(input.candidateActions)}`,
|
|
1948
|
+
input.observedAction ? `observedAction: ${JSON.stringify(input.observedAction)}` : "",
|
|
1949
|
+
input.context ? `context:
|
|
1950
|
+
${input.context}` : "",
|
|
1951
|
+
"",
|
|
1952
|
+
"Schema:",
|
|
1953
|
+
JSON.stringify({
|
|
1954
|
+
predictedAction: "one candidate action",
|
|
1955
|
+
confidence: "number in [0,1]",
|
|
1956
|
+
beliefSummary: "short outcome-blind summary",
|
|
1957
|
+
uncertainty: ["short uncertainty"],
|
|
1958
|
+
evidenceRefs: ["evidence id"],
|
|
1959
|
+
wouldChangeMindIf: ["observable evidence"],
|
|
1960
|
+
targetProb: "optional number in [0,1]",
|
|
1961
|
+
qHat: "optional number in [0,1]"
|
|
1962
|
+
})
|
|
1963
|
+
].filter(Boolean).join("\n");
|
|
1964
|
+
}
|
|
1965
|
+
async function probePoint(point, options) {
|
|
1966
|
+
const diagnostics = [];
|
|
1967
|
+
const candidateActions = uniqueStrings2(point.candidateActions ?? []);
|
|
1968
|
+
if ((options.requireCandidateActions ?? true) && candidateActions.length === 0) {
|
|
1969
|
+
diagnostics.push({
|
|
1970
|
+
decisionId: point.id,
|
|
1971
|
+
severity: "warning",
|
|
1972
|
+
reason: "missing candidateActions"
|
|
1973
|
+
});
|
|
1974
|
+
return { diagnostics };
|
|
1975
|
+
}
|
|
1976
|
+
let response;
|
|
1977
|
+
try {
|
|
1978
|
+
response = await options.probe({
|
|
1979
|
+
probeId: options.probeId,
|
|
1980
|
+
decisionId: point.id,
|
|
1981
|
+
runId: point.runId,
|
|
1982
|
+
scenarioId: point.scenarioId,
|
|
1983
|
+
stepIndex: point.stepIndex,
|
|
1984
|
+
decisionKind: point.kind,
|
|
1985
|
+
candidateActions,
|
|
1986
|
+
...options.includeObservedAction ? { observedAction: point.chosenAction } : {},
|
|
1987
|
+
evidence: point.evidence.map((ref) => ({
|
|
1988
|
+
id: ref.id,
|
|
1989
|
+
source: ref.source,
|
|
1990
|
+
...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
|
|
1991
|
+
...ref.quality ? { quality: ref.quality } : {}
|
|
1992
|
+
})),
|
|
1993
|
+
context: trimText2(await options.contextOf?.(point), options.maxContextChars),
|
|
1994
|
+
metadata: await options.metadataOf?.(point)
|
|
1995
|
+
});
|
|
1996
|
+
} catch (error) {
|
|
1997
|
+
diagnostics.push({
|
|
1998
|
+
decisionId: point.id,
|
|
1999
|
+
severity: "error",
|
|
2000
|
+
reason: `probe threw: ${errorMessage2(error)}`
|
|
2001
|
+
});
|
|
2002
|
+
return { diagnostics };
|
|
2003
|
+
}
|
|
2004
|
+
const normalized = normalizeProbeResponse(response, {
|
|
2005
|
+
point,
|
|
2006
|
+
candidateActions,
|
|
2007
|
+
allowOutOfSetActions: options.allowOutOfSetActions ?? false
|
|
2008
|
+
});
|
|
2009
|
+
if (!normalized.record) {
|
|
2010
|
+
diagnostics.push(...normalized.diagnostics);
|
|
2011
|
+
return { diagnostics };
|
|
2012
|
+
}
|
|
2013
|
+
return {
|
|
2014
|
+
record: {
|
|
2015
|
+
probeId: options.probeId,
|
|
2016
|
+
decisionId: point.id,
|
|
2017
|
+
runId: point.runId,
|
|
2018
|
+
scenarioId: point.scenarioId,
|
|
2019
|
+
stepIndex: point.stepIndex,
|
|
2020
|
+
decisionKind: point.kind,
|
|
2021
|
+
candidateActions,
|
|
2022
|
+
observedAction: point.chosenAction,
|
|
2023
|
+
agreesWithObservedAction: normalized.record.predictedAction === point.chosenAction,
|
|
2024
|
+
...options.includeOutcomeInRecord === false ? {} : { outcome: point.outcome },
|
|
2025
|
+
...normalized.record
|
|
2026
|
+
},
|
|
2027
|
+
diagnostics
|
|
2028
|
+
};
|
|
2029
|
+
}
|
|
2030
|
+
function normalizeProbeResponse(response, options) {
|
|
2031
|
+
const diagnostics = [];
|
|
2032
|
+
const predictedAction = stringOrNull(response.predictedAction);
|
|
2033
|
+
if (!predictedAction) {
|
|
2034
|
+
diagnostics.push({
|
|
2035
|
+
decisionId: options.point.id,
|
|
2036
|
+
severity: "error",
|
|
2037
|
+
reason: "missing predictedAction"
|
|
2038
|
+
});
|
|
2039
|
+
} else if (!options.allowOutOfSetActions && options.candidateActions.length > 0 && !options.candidateActions.includes(predictedAction)) {
|
|
2040
|
+
diagnostics.push({
|
|
2041
|
+
decisionId: options.point.id,
|
|
2042
|
+
severity: "error",
|
|
2043
|
+
reason: `predictedAction ${predictedAction} is not in candidateActions`
|
|
2044
|
+
});
|
|
2045
|
+
}
|
|
2046
|
+
if (!isUnitProbability(response.confidence)) {
|
|
2047
|
+
diagnostics.push({
|
|
2048
|
+
decisionId: options.point.id,
|
|
2049
|
+
severity: "error",
|
|
2050
|
+
reason: `invalid confidence ${String(response.confidence)}`
|
|
2051
|
+
});
|
|
2052
|
+
}
|
|
2053
|
+
if (response.targetProb !== void 0 && !isUnitProbability(response.targetProb)) {
|
|
2054
|
+
diagnostics.push({
|
|
2055
|
+
decisionId: options.point.id,
|
|
2056
|
+
severity: "error",
|
|
2057
|
+
reason: `invalid targetProb ${String(response.targetProb)}`
|
|
2058
|
+
});
|
|
2059
|
+
}
|
|
2060
|
+
if (response.qHat !== void 0 && response.qHat !== null && !isUnitProbability(response.qHat)) {
|
|
2061
|
+
diagnostics.push({
|
|
2062
|
+
decisionId: options.point.id,
|
|
2063
|
+
severity: "error",
|
|
2064
|
+
reason: `invalid qHat ${String(response.qHat)}`
|
|
2065
|
+
});
|
|
2066
|
+
}
|
|
2067
|
+
if (diagnostics.length > 0 || !predictedAction) return { diagnostics };
|
|
2068
|
+
return {
|
|
2069
|
+
record: {
|
|
2070
|
+
predictedAction,
|
|
2071
|
+
confidence: response.confidence,
|
|
2072
|
+
...response.beliefSummary ? { beliefSummary: trimText2(response.beliefSummary, 2e3) } : {},
|
|
2073
|
+
uncertainty: compactStrings(response.uncertainty),
|
|
2074
|
+
evidenceRefs: compactStrings(response.evidenceRefs),
|
|
2075
|
+
wouldChangeMindIf: compactStrings(response.wouldChangeMindIf),
|
|
2076
|
+
...response.targetProb !== void 0 ? { targetProb: response.targetProb } : {},
|
|
2077
|
+
...response.qHat !== void 0 ? { qHat: response.qHat } : {},
|
|
2078
|
+
...response.metadata ? { metadata: response.metadata } : {}
|
|
2079
|
+
},
|
|
2080
|
+
diagnostics
|
|
2081
|
+
};
|
|
2082
|
+
}
|
|
2083
|
+
function summarizeShadowProbe(attempted, records) {
|
|
2084
|
+
const confidences = records.map((record2) => record2.confidence);
|
|
2085
|
+
const agreements = records.filter((record2) => record2.agreesWithObservedAction).length;
|
|
2086
|
+
return {
|
|
2087
|
+
attempted,
|
|
2088
|
+
completed: records.length,
|
|
2089
|
+
dropped: attempted - records.length,
|
|
2090
|
+
withOutcome: records.filter((record2) => record2.outcome !== void 0).length,
|
|
2091
|
+
withTargetProb: records.filter((record2) => record2.targetProb !== void 0).length,
|
|
2092
|
+
meanConfidence: confidences.length > 0 ? mean3(confidences) : null,
|
|
2093
|
+
observedAgreementRate: records.length > 0 ? agreements / records.length : null
|
|
2094
|
+
};
|
|
2095
|
+
}
|
|
2096
|
+
function isUnitProbability(value) {
|
|
2097
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
|
|
2098
|
+
}
|
|
2099
|
+
function boundedInteger(value, min, max) {
|
|
2100
|
+
if (!Number.isFinite(value)) return min;
|
|
2101
|
+
return Math.max(min, Math.min(max, Math.floor(value)));
|
|
2102
|
+
}
|
|
2103
|
+
function compactStrings(values, maxItems = 12) {
|
|
2104
|
+
if (!Array.isArray(values)) return [];
|
|
2105
|
+
return values.filter((value) => typeof value === "string" && value.length > 0).slice(0, maxItems).map((value) => trimText2(value, 500) ?? "").filter(Boolean);
|
|
2106
|
+
}
|
|
2107
|
+
function uniqueStrings2(values) {
|
|
2108
|
+
return [...new Set(values.filter((value) => value.length > 0))];
|
|
2109
|
+
}
|
|
2110
|
+
function stringOrNull(value) {
|
|
2111
|
+
return typeof value === "string" && value.length > 0 ? value : null;
|
|
2112
|
+
}
|
|
2113
|
+
function trimText2(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS2) {
|
|
2114
|
+
if (!value) return void 0;
|
|
2115
|
+
return value.length > maxChars ? value.slice(value.length - maxChars) : value;
|
|
2116
|
+
}
|
|
2117
|
+
function mean3(values) {
|
|
2118
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
2119
|
+
}
|
|
2120
|
+
function errorMessage2(error) {
|
|
2121
|
+
return error instanceof Error ? error.message : String(error);
|
|
2122
|
+
}
|
|
2123
|
+
export {
|
|
2124
|
+
BELIEF_DECISION_KINDS,
|
|
2125
|
+
BELIEF_EVALUATION_CRITERIA,
|
|
2126
|
+
BELIEF_EVIDENCE_QUALITIES,
|
|
2127
|
+
BELIEF_EVIDENCE_SOURCES,
|
|
2128
|
+
analyzeBeliefDecisionCorpus,
|
|
2129
|
+
analyzeBeliefPolicy,
|
|
2130
|
+
beliefDecisionsToOffPolicyTrajectories,
|
|
2131
|
+
buildBeliefDecisionResearchEvidencePacket,
|
|
2132
|
+
buildCodeAgentBeliefEvidenceCorpus,
|
|
2133
|
+
buildRuntimeBeliefPhase0Measurement,
|
|
2134
|
+
buildRuntimeBenchmarkBeliefPhase0Measurement,
|
|
2135
|
+
calibrateBeliefDecisions,
|
|
2136
|
+
createBeliefRuntimeHookCollector,
|
|
2137
|
+
embeddedBeliefOpeTargetPolicy,
|
|
2138
|
+
evaluateBeliefOffPolicy,
|
|
2139
|
+
evaluateBeliefSelectivePolicy,
|
|
2140
|
+
extractBeliefDecisionPoints,
|
|
2141
|
+
extractCodeAgentBeliefDecisionPoints,
|
|
2142
|
+
formatBeliefShadowProbePrompt,
|
|
2143
|
+
inventoryBeliefDecisionPoints,
|
|
2144
|
+
isBeliefDecisionKind,
|
|
2145
|
+
isBeliefEvidenceSource,
|
|
2146
|
+
runBeliefShadowProbe,
|
|
2147
|
+
runtimeDecisionPointToBeliefDecisionPoint,
|
|
2148
|
+
runtimeDecisionPointToBeliefShadowProbeInput,
|
|
2149
|
+
selectBeliefDecisionTarget,
|
|
2150
|
+
thresholdSelectivePolicy
|
|
2151
|
+
};
|
|
2152
|
+
//# sourceMappingURL=index.js.map
|