@tangle-network/agent-eval 0.120.1 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +6 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4972
  125. package/dist/contract/index.js +0 -1654
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,1301 +0,0 @@
1
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
2
- interface BudgetSpec {
3
- tokens?: number;
4
- wallMs?: number;
5
- calls?: number;
6
- usd?: number;
7
- }
8
- interface RunOutcome$1 {
9
- score?: number;
10
- pass?: boolean;
11
- failureClass?: FailureClass;
12
- notes?: string;
13
- }
14
- /**
15
- * Layer — optional classification in a nested build workflow.
16
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
17
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
18
- * `app-runtime`: a run of the generated agent against a domain scenario.
19
- * `meta`: any meta-eval (judge replay, correlation analysis).
20
- */
21
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
22
- interface Run {
23
- runId: string;
24
- /**
25
- * Stable identifier of the scenario being executed.
26
- *
27
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
28
- * input WITHOUT this field, substituting a sensible default
29
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
30
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
31
- * keeps the persisted shape unambiguous for downstream filters + aggregations
32
- * while removing the boilerplate of inventing placeholder ids at the call site.
33
- */
34
- scenarioId: string;
35
- variantId?: string;
36
- datasetVersion?: string;
37
- /** Git SHA of agent code at run time. */
38
- codeSha?: string;
39
- /** Hash of the prompt template + any system prompt. */
40
- promptSha?: string;
41
- /** Model id + date + system-prompt hash, concatenated. */
42
- modelFingerprint?: string;
43
- seed?: number;
44
- /** Arbitrary environment markers (shell, docker version, tz). */
45
- envFingerprint?: Record<string, string>;
46
- /** Version of the redaction rules applied to this run. */
47
- redactionVersion?: string;
48
- /** Parent run in a nested build workflow. A builder run's children are
49
- * app-build runs; those children are app-runtime runs. */
50
- parentRunId?: string;
51
- /** Stable project identifier — groups runs across chats + sessions. */
52
- projectId?: string;
53
- /** Chat/conversation identifier within a project. */
54
- chatId?: string;
55
- /** Layer classification — hint for aggregation; not enforced. */
56
- layer?: RunLayer;
57
- startedAt: number;
58
- endedAt?: number;
59
- status: RunStatus;
60
- outcome?: RunOutcome$1;
61
- budget?: BudgetSpec;
62
- /** Free-form labels for downstream grouping. */
63
- tags?: Record<string, string>;
64
- }
65
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
66
- type SpanStatus = 'ok' | 'error';
67
- interface SpanBase {
68
- spanId: string;
69
- parentSpanId?: string;
70
- runId: string;
71
- kind: SpanKind;
72
- name: string;
73
- startedAt: number;
74
- endedAt?: number;
75
- status?: SpanStatus;
76
- error?: string;
77
- /** Anything not covered by typed fields. Kept deliberately free-form. */
78
- attributes?: Record<string, unknown>;
79
- }
80
- interface Message {
81
- role: 'system' | 'user' | 'assistant' | 'tool';
82
- content: string;
83
- tokens?: number;
84
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
85
- images?: Array<{
86
- artifactId?: string;
87
- url?: string;
88
- mime?: string;
89
- }>;
90
- }
91
- interface LlmSpan extends SpanBase {
92
- kind: 'llm';
93
- model: string;
94
- messages: Message[];
95
- output?: string;
96
- inputTokens?: number;
97
- /** All generated tokens, including the reasoning subset when present. */
98
- outputTokens?: number;
99
- cachedTokens?: number;
100
- cacheWriteTokens?: number;
101
- /** Reasoning-token subset of `outputTokens`. */
102
- reasoningTokens?: number;
103
- costUsd?: number;
104
- finishReason?: string;
105
- }
106
- interface ToolSpan extends SpanBase {
107
- kind: 'tool';
108
- toolName: string;
109
- args: unknown;
110
- /** False when the source observed the call but did not capture its arguments. */
111
- argsCaptured?: boolean;
112
- result?: unknown;
113
- latencyMs?: number;
114
- }
115
- interface RetrievalSpan extends SpanBase {
116
- kind: 'retrieval';
117
- query: string;
118
- hits: Array<{
119
- docId: string;
120
- score: number;
121
- content?: string;
122
- }>;
123
- }
124
- interface JudgeSpan extends SpanBase {
125
- kind: 'judge';
126
- judgeId: string;
127
- /** Span this judgment applies to. */
128
- targetSpanId: string;
129
- dimension: string;
130
- /** Numeric score (free-range; interpretation up to the judge). */
131
- score: number;
132
- rationale?: string;
133
- evidence?: string;
134
- }
135
- interface SandboxSpan extends SpanBase {
136
- kind: 'sandbox';
137
- image?: string;
138
- command?: string;
139
- exitCode?: number;
140
- testsTotal?: number;
141
- testsPassed?: number;
142
- stdoutHash?: string;
143
- stderrHash?: string;
144
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
145
- wallMs?: number;
146
- }
147
- interface GenericSpan extends SpanBase {
148
- kind: 'agent' | 'custom';
149
- }
150
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
151
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
152
- interface TraceEvent {
153
- eventId: string;
154
- runId: string;
155
- spanId?: string;
156
- kind: EventKind;
157
- timestamp: number;
158
- payload: Record<string, unknown>;
159
- }
160
- interface BudgetLedgerEntry {
161
- runId: string;
162
- dimension: keyof BudgetSpec;
163
- limit: number;
164
- consumed: number;
165
- remaining: number;
166
- timestamp: number;
167
- breached: boolean;
168
- /** Span that triggered this entry, if any. */
169
- spanId?: string;
170
- }
171
- interface Artifact {
172
- artifactId: string;
173
- runId: string;
174
- spanId?: string;
175
- contentType: string;
176
- sizeBytes: number;
177
- /** sha256 in hex. */
178
- hash: string;
179
- /** External storage URL (R2, S3, filesystem path). */
180
- storageUrl?: string;
181
- /** Inline content for small blobs — keep under ~64KB. */
182
- inlineContent?: string;
183
- }
184
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
185
-
186
- interface RunFilter {
187
- scenarioId?: string;
188
- variantId?: string;
189
- status?: RunStatus;
190
- since?: number;
191
- until?: number;
192
- tag?: {
193
- key: string;
194
- value: string;
195
- };
196
- parentRunId?: string;
197
- projectId?: string;
198
- chatId?: string;
199
- layer?: RunLayer;
200
- }
201
- interface SpanFilter {
202
- runId?: string;
203
- parentSpanId?: string;
204
- kind?: SpanKind;
205
- name?: string;
206
- toolName?: string;
207
- judgeId?: string;
208
- since?: number;
209
- until?: number;
210
- }
211
- interface EventFilter {
212
- runId?: string;
213
- spanId?: string;
214
- kind?: EventKind;
215
- since?: number;
216
- until?: number;
217
- }
218
- interface TraceStore {
219
- appendRun(run: Run): Promise<void>;
220
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
221
- appendSpan(span: Span): Promise<void>;
222
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
223
- appendEvent(event: TraceEvent): Promise<void>;
224
- appendArtifact(artifact: Artifact): Promise<void>;
225
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
226
- getRun(runId: string): Promise<Run | undefined>;
227
- listRuns(filter?: RunFilter): Promise<Run[]>;
228
- spans(filter?: SpanFilter): Promise<Span[]>;
229
- events(filter?: EventFilter): Promise<TraceEvent[]>;
230
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
231
- artifacts(runId: string): Promise<Artifact[]>;
232
- }
233
-
234
- /**
235
- * Calibration curve — binned "if eval says X, what does reality show?"
236
- *
237
- * Companion to correlationStudy. Raw correlation is a single number;
238
- * the calibration curve shows *where* the eval is well-calibrated vs
239
- * overconfident / underconfident. Buckets the eval metric, computes
240
- * mean outcome per bucket, reports expected-calibration-error (ECE).
241
- */
242
-
243
- interface CalibrationBin {
244
- lower: number;
245
- upper: number;
246
- n: number;
247
- evalMean: number;
248
- outcomeMean: number;
249
- /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */
250
- gap: number;
251
- }
252
- interface CalibrationReport {
253
- evalMetric: string;
254
- outcomeMetric: string;
255
- n: number;
256
- bins: CalibrationBin[];
257
- /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */
258
- ece: number;
259
- /** Max bin gap — upper bound on miscalibration. */
260
- maxGap: number;
261
- }
262
-
263
- /**
264
- * Off-policy evaluation primitives.
265
- *
266
- * Standard inverse-probability-weighted (IPS), self-normalized
267
- * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
268
- * value of a *target* policy given trajectories collected under a
269
- * *behavior* policy. This is the canonical RL eval task: "we have last
270
- * week's runs, we changed the policy — how would the new one do without
271
- * re-running?"
272
- *
273
- * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
274
- * & Joachims 2015 for SNIPS) but the *application* to LLM-agent
275
- * evaluation needs care:
276
- *
277
- * - The "policy" is the (prompt, tool config, model snapshot) triple.
278
- * Two policies have the same probability over an action *iff* their
279
- * LLM call would emit the same token with the same probability —
280
- * which is generally unknowable without the model log-probs.
281
- * - For LLM agents, propensity scores must be supplied by the caller
282
- * (logged in the trace, recovered from token log-probs, or estimated
283
- * via a learned propensity model). We do NOT estimate propensity here.
284
- * - Doubly-robust requires a Q-function (model-based reward predictor).
285
- * We accept any callable; consumers pass either a tabular average,
286
- * a regression fit, or a learned reward model.
287
- *
288
- * Bias / variance tradeoffs:
289
- * - IPS: unbiased; high variance for small overlap, infinite variance
290
- * when target has support outside behavior.
291
- * - SNIPS: lower variance, slight bias; usually preferred in practice.
292
- * - DR: doubly-robust — unbiased if either propensity OR Q-function is
293
- * correct. Lowest practical variance when Q is decent. Use this.
294
- *
295
- * Caveat the panel will land: on the LLM-agent setting, propensity scores
296
- * recovered from token log-probs are noisy, the action space is enormous,
297
- * and overlap is often poor. These estimators are useful but not magic;
298
- * complement with `replayCampaign` (exact replay where the request hashes
299
- * match) for high-confidence answers and OPE for the gap.
300
- */
301
- interface OffPolicyTrajectory {
302
- /** Stable id, for traceability through the dataset. */
303
- runId: string;
304
- /** Reward observed under the behavior policy (the realized outcome). */
305
- reward: number;
306
- /**
307
- * Behavior-policy probability of the action that was taken. For LLM
308
- * agents this is typically `exp(sum(token_log_probs))` over the chosen
309
- * trajectory. Must be in (0, 1].
310
- */
311
- behaviorProb: number;
312
- /**
313
- * Target-policy probability of the same action. For replay-style
314
- * counterfactual evaluation this is what the *new* policy would have
315
- * assigned to the *old* trajectory. Must be in [0, 1].
316
- */
317
- targetProb: number;
318
- /**
319
- * Optional model-based reward prediction at the same context. Used by
320
- * `doublyRobust`. Set to `null` for IPS-only evaluation.
321
- */
322
- qHat?: number | null;
323
- }
324
- interface OffPolicyEstimate {
325
- /** Estimated value of the target policy. */
326
- value: number;
327
- /** Standard error of the estimate. */
328
- standardError: number;
329
- /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
330
- effectiveSampleSize: number;
331
- /** Number of trajectories used. */
332
- n: number;
333
- /**
334
- * Diagnostic: maximum importance weight observed. Large values (>>10x
335
- * mean) are a red flag — variance is dominated by a few outliers.
336
- */
337
- maxImportanceWeight: number;
338
- }
339
- interface OffPolicyOptions {
340
- /**
341
- * Cap importance weights at this value (Ionides 2008 truncated IS) to
342
- * trade unbiasedness for variance reduction. Default `Infinity` (no cap).
343
- * Set e.g. `10` for stable estimates when the policies are close.
344
- */
345
- weightCap?: number;
346
- /** Reward clipping range. Default `[0, 1]`. */
347
- rewardClip?: {
348
- low: number;
349
- high: number;
350
- };
351
- }
352
-
353
- declare const BELIEF_DECISION_KINDS: readonly ["continue", "verify", "ask", "retry", "stop", "memory-write", "memory-read", "tool-select", "skill-select", "workflow-select", "surface-promote"];
354
- type BeliefDecisionKind = (typeof BELIEF_DECISION_KINDS)[number];
355
- declare const BELIEF_EVIDENCE_SOURCES: readonly ["run", "span", "event", "finding", "memory", "knowledge", "policy"];
356
- type BeliefEvidenceSource = (typeof BELIEF_EVIDENCE_SOURCES)[number];
357
- declare const BELIEF_EVIDENCE_QUALITIES: readonly ["direct", "derived", "self-reported", "unverified", "stale", "contradicted"];
358
- type BeliefEvidenceQuality = (typeof BELIEF_EVIDENCE_QUALITIES)[number];
359
- declare const BELIEF_EVALUATION_CRITERIA: readonly [{
360
- readonly id: "capture-integrity";
361
- readonly label: "Capture integrity";
362
- readonly reasonCodes: readonly ["trace-missing", "run-record-missing", "backend-integrity-missing"];
363
- }, {
364
- readonly id: "decision-completeness";
365
- readonly label: "Decision completeness";
366
- readonly reasonCodes: readonly ["candidate-actions-missing", "chosen-action-missing", "decision-evidence-missing"];
367
- }, {
368
- readonly id: "evidence-quality";
369
- readonly label: "Evidence quality";
370
- readonly reasonCodes: readonly ["evidence-stale", "evidence-contradictory", "evidence-unverified", "evidence-self-reported"];
371
- }, {
372
- readonly id: "outcome-quality";
373
- readonly label: "Outcome quality";
374
- readonly reasonCodes: readonly ["outcome-missing", "outcome-delayed", "cost-missing"];
375
- }, {
376
- readonly id: "calibration";
377
- readonly label: "Calibration";
378
- readonly reasonCodes: readonly ["confidence-missing", "calibration-unsupported", "calibration-gap-high"];
379
- }, {
380
- readonly id: "accepted-region-risk";
381
- readonly label: "Accepted-region risk";
382
- readonly reasonCodes: readonly ["accepted-error-high", "coverage-too-low"];
383
- }, {
384
- readonly id: "policy-value";
385
- readonly label: "Policy value";
386
- readonly reasonCodes: readonly ["utility-lift-missing", "baseline-dominates", "cost-too-high"];
387
- }, {
388
- readonly id: "ope-support";
389
- readonly label: "OPE support";
390
- readonly reasonCodes: readonly ["behavior-propensity-missing", "behavior-propensity-invalid", "target-propensity-missing", "target-propensity-invalid", "effective-sample-size-low", "importance-weight-high"];
391
- }, {
392
- readonly id: "memory-health";
393
- readonly label: "Memory health";
394
- readonly reasonCodes: readonly ["memory-stale", "memory-poisoning-risk", "context-bloat", "memory-write-unverified"];
395
- }, {
396
- readonly id: "surface-attribution";
397
- readonly label: "Surface attribution";
398
- readonly reasonCodes: readonly ["surface-claim-unsupported", "causal-attribution-missing"];
399
- }, {
400
- readonly id: "generalization";
401
- readonly label: "Generalization";
402
- readonly reasonCodes: readonly ["split-missing", "holdout-regression", "task-family-coverage-low", "leakage-risk"];
403
- }, {
404
- readonly id: "promotion";
405
- readonly label: "Promotion";
406
- readonly reasonCodes: readonly ["negative-control-failed", "promotion-gate-failed", "human-review-required"];
407
- }];
408
- type BeliefEvaluationCriterionId = (typeof BELIEF_EVALUATION_CRITERIA)[number]['id'];
409
- type BeliefDecisionReasonCode = (typeof BELIEF_EVALUATION_CRITERIA)[number]['reasonCodes'][number];
410
- interface BeliefDecisionReason {
411
- code: BeliefDecisionReasonCode;
412
- criterion?: BeliefEvaluationCriterionId;
413
- detail?: string;
414
- evidenceIds?: string[];
415
- metadata?: Record<string, unknown>;
416
- }
417
- declare function isBeliefDecisionKind(value: unknown): value is BeliefDecisionKind;
418
- declare function isBeliefEvidenceSource(value: unknown): value is BeliefEvidenceSource;
419
- interface BeliefEvidenceRef {
420
- source: BeliefEvidenceSource;
421
- id: string;
422
- runId?: string;
423
- spanId?: string;
424
- eventId?: string;
425
- detail?: string;
426
- quality?: BeliefEvidenceQuality;
427
- observedAt?: string;
428
- metadata?: Record<string, unknown>;
429
- }
430
- interface BeliefDecisionOutcome {
431
- success?: boolean;
432
- score?: number;
433
- reward?: number;
434
- costUsd?: number;
435
- observedAt?: string;
436
- metadata?: Record<string, unknown>;
437
- }
438
- interface BeliefDecisionPoint {
439
- id: string;
440
- runId: string;
441
- scenarioId?: string;
442
- stepIndex: number;
443
- kind: BeliefDecisionKind;
444
- chosenAction: string;
445
- candidateActions?: string[];
446
- confidence?: number;
447
- behaviorProb?: number;
448
- targetProb?: number;
449
- qHat?: number | null;
450
- costUsd?: number;
451
- evidence: BeliefEvidenceRef[];
452
- outcome?: BeliefDecisionOutcome;
453
- reasons?: BeliefDecisionReason[];
454
- metadata?: Record<string, unknown>;
455
- }
456
- interface BeliefDecisionExtractionDiagnostic {
457
- runId: string;
458
- eventId?: string;
459
- severity: 'info' | 'warning' | 'error';
460
- reason: string;
461
- }
462
- interface BeliefDecisionExtractionReport {
463
- decisions: BeliefDecisionPoint[];
464
- diagnostics: BeliefDecisionExtractionDiagnostic[];
465
- }
466
- type BeliefPolicyAction = 'accept' | 'defer' | 'verify' | 'ask' | 'retry' | 'stop';
467
- interface BeliefPolicyDecision {
468
- action: BeliefPolicyAction;
469
- confidence?: number;
470
- targetProb?: number;
471
- qHat?: number | null;
472
- reason?: string;
473
- reasons?: BeliefDecisionReason[];
474
- }
475
- interface BeliefSelectivePolicy {
476
- id: string;
477
- decide(point: BeliefDecisionPoint): BeliefPolicyDecision;
478
- }
479
- interface BeliefOpeTargetPolicy {
480
- id: string;
481
- targetProbOf(point: BeliefDecisionPoint): number | null | undefined;
482
- qHatOf?(point: BeliefDecisionPoint): number | null | undefined;
483
- }
484
- interface BeliefUtilityOptions {
485
- successUtility?: number;
486
- failureUtility?: number;
487
- deferUtility?: number;
488
- verifyCost?: number;
489
- askCost?: number;
490
- retryCost?: number;
491
- stopUtility?: number;
492
- costWeight?: number;
493
- }
494
- interface BeliefSelectivePolicyMetrics {
495
- policyId: string;
496
- n: number;
497
- accepted: number;
498
- rejected: number;
499
- coverage: number;
500
- acceptedErrorRate: number;
501
- baselineUtility: number;
502
- policyUtility: number;
503
- utilityDelta: number;
504
- utilityCi95: {
505
- mean: number;
506
- lower: number;
507
- upper: number;
508
- };
509
- rejectedMeanReward: number | null;
510
- recommendation: 'ship' | 'hold' | 'need_more_data';
511
- reasons: string[];
512
- }
513
- interface BeliefOpeSupportDiagnostics {
514
- supported: boolean;
515
- n: number;
516
- dropped: number;
517
- effectiveSampleSize: number;
518
- effectiveSampleRatio: number;
519
- maxImportanceWeight: number;
520
- reasons: string[];
521
- }
522
- interface BeliefOpeReport {
523
- targetPolicyId: string;
524
- ips: OffPolicyEstimate;
525
- snips: OffPolicyEstimate;
526
- dr: OffPolicyEstimate;
527
- support: BeliefOpeSupportDiagnostics;
528
- }
529
- type BeliefEvaluationStatus = 'ship' | 'hold' | 'need_more_data';
530
- type BeliefCalibrationStatus = 'supported' | 'unsupported';
531
- type BeliefOpeStatus = 'supported' | 'unsupported' | 'not_requested';
532
- interface BeliefPolicyEvaluationReport {
533
- policyId: string;
534
- n: number;
535
- status: BeliefEvaluationStatus;
536
- selectiveStatus: BeliefEvaluationStatus;
537
- calibrationStatus: BeliefCalibrationStatus;
538
- opeStatus: BeliefOpeStatus;
539
- opeTargetPolicyId?: string;
540
- selective: BeliefSelectivePolicyMetrics;
541
- calibration?: CalibrationReport;
542
- ope?: BeliefOpeReport;
543
- diagnostics: string[];
544
- }
545
-
546
- type BeliefCalibrationRegion = 'all' | 'accepted' | 'rejected';
547
- interface BeliefCalibrationOptions {
548
- bins?: number;
549
- minPairs?: number;
550
- policy?: BeliefSelectivePolicy;
551
- region?: BeliefCalibrationRegion;
552
- }
553
- declare function calibrateBeliefDecisions(points: BeliefDecisionPoint[], options?: BeliefCalibrationOptions): CalibrationReport | null;
554
-
555
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
556
- type AgentProfileDimensionValue = string | number | boolean | null;
557
- interface AgentProfileSource {
558
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
559
- kind: string;
560
- /** sha256 over the canonical source profile object. */
561
- hash: string;
562
- }
563
- interface AgentProfileHarness {
564
- id: string;
565
- version?: string;
566
- hash?: string;
567
- }
568
- interface AgentProfileCell {
569
- schemaVersion: AgentProfileCellSchemaVersion;
570
- cellId: string;
571
- profileId: string;
572
- sourceProfile: AgentProfileSource;
573
- harness?: AgentProfileHarness;
574
- model?: string;
575
- promptHash?: string;
576
- dimensions?: Record<string, AgentProfileDimensionValue>;
577
- }
578
-
579
- /**
580
- * Paper-grade RunRecord schema + runtime validator.
581
- *
582
- * Every run that participates in a promotion gate, paper table, or
583
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
584
- * fields are exactly those the paper "Two Loops, Three Roles" requires
585
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
586
- * holdout split tag and either a `searchScore` or a `holdoutScore`.
587
- *
588
- * This is intentionally NOT a replacement for the rich `Run` /
589
- * `ProposeReviewReport` / `ScenarioResult` types already in the
590
- * package. Those are runtime structures with full provenance. A
591
- * `RunRecord` is the analysis-time projection — the JSON-friendly
592
- * row you'd put in a parquet file or paste into a notebook.
593
- *
594
- * Validate at the boundary:
595
- *
596
- * const rec = validateRunRecord(rawJson) // throws on missing
597
- * const ok = isRunRecord(rawJson) // boolean check
598
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
599
- *
600
- * The validator runs in pure TS — zod is intentionally NOT a
601
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
602
- */
603
-
604
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
605
- * combined train+test pool that the optimizer is allowed to read. */
606
- type RunSplitTag = 'search' | 'dev' | 'holdout';
607
- interface RunTokenUsage {
608
- input: number;
609
- /** All generated tokens charged as output, including reasoning tokens. */
610
- output: number;
611
- /** Reasoning-token subset of `output`, when the provider reports it. */
612
- reasoning?: number;
613
- /** Prompt tokens served from a provider cache. */
614
- cached?: number;
615
- /** Prompt tokens written into a provider cache. */
616
- cacheWrite?: number;
617
- }
618
- /**
619
- * How a run's USD amount was obtained.
620
- *
621
- * `costUsd` remains mandatory for wire compatibility. New producers should
622
- * always populate this discriminated union so a missing bill is never
623
- * mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
624
- * the legacy `0` sentinel while this field carries the truthful null.
625
- */
626
- type RunCostProvenance = {
627
- kind: 'observed';
628
- usd: number;
629
- } | {
630
- kind: 'estimated';
631
- usd: number;
632
- } | {
633
- kind: 'uncaptured';
634
- usd: null;
635
- };
636
- interface RunJudgeMetadata {
637
- model: string;
638
- promptVersion: string;
639
- /** [0,1] confidence the judge declared. Constant judge confidence
640
- * across many runs is a fallback signal (see `canary.ts`). */
641
- confidence: number;
642
- /** True if the judge degraded to a fallback path (rules-only,
643
- * prior-call cache, etc.). The canary uses this to alert. */
644
- fallback: boolean;
645
- }
646
- /**
647
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
648
- * judges over a multi-dimensional rubric.
649
- *
650
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
651
- * composite the gate uses. The full breakdown belongs here so consumers
652
- * can answer "which judge disagreed?", "which dimension dragged the
653
- * composite down?", and "did half the panel fail?" without re-running.
654
- *
655
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
656
- * `composite` are convenience projections — derivable but precomputed so
657
- * downstream IRR primitives (`interRaterReliability`,
658
- * `corpusInterRaterAgreement`) and reporters don't pay the same
659
- * aggregation twice.
660
- *
661
- * Fail-loud discipline: judges that errored out land in `failedJudges`
662
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
663
- * run); the explicit list makes a partial-failure recorded as such.
664
- */
665
- interface JudgeScoresRecord {
666
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
667
- perJudge: Record<string, Record<string, number>>;
668
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
669
- perDimMean: Record<string, number>;
670
- /** Composite mean across all dims and judges. Mirrors the score
671
- * the gate sees on `outcome.searchScore` / `holdoutScore`. */
672
- composite: number;
673
- /** Judges that errored or returned an unparseable verdict. Recorded
674
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
675
- * not inferred from missing keys in `perJudge`. */
676
- failedJudges?: string[];
677
- /** Free-form notes the judges emitted (joined across judges or
678
- * first-judge only — consumer's choice). */
679
- notes?: string;
680
- }
681
- interface RunOutcome {
682
- /** Score on the search/optimization split. Optional because a
683
- * holdout-only evaluation only fills `holdoutScore`. */
684
- searchScore?: number;
685
- /** Score on the held-out split. Optional because a search-only run
686
- * only fills `searchScore`. At least one must be present. */
687
- holdoutScore?: number;
688
- /** Bag of any other metric the run produced — judge dimensions,
689
- * pass/fail counters, latency stats, etc. Numeric only — keeps
690
- * reporters honest. */
691
- raw: Record<string, number>;
692
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
693
- * judgements populate this; substrate primitives like
694
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
695
- * these records as input. Optional — single-judge or scalar-only
696
- * runs leave it unset. */
697
- judgeScores?: JudgeScoresRecord;
698
- /** Authenticity / realness verdict — did the run build the REAL thing on the
699
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
700
- * with an authenticity config populate it. Carried in the corpus so the
701
- * flywheel / off-policy learning can optimize for real completion, not gamed
702
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
703
- * must not count as a real success regardless of `score`. */
704
- realness?: {
705
- score: number;
706
- gated: boolean;
707
- reason?: string;
708
- };
709
- }
710
- /**
711
- * Mandatory paper-grade fields for a single evaluation run. Optional
712
- * fields are extension points; mandatory fields throw if missing.
713
- *
714
- * Hash discipline:
715
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
716
- * model (after any steering bundle merge).
717
- * - `configHash` is the sha256 of the effective run config (model,
718
- * temperature, tools, judges, splits). The pair (promptHash,
719
- * configHash) uniquely identifies an experiment cell.
720
- *
721
- * Model snapshot discipline:
722
- * - `model` MUST encode a snapshot version. Bare aliases like
723
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
724
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
725
- */
726
- interface RunRecord {
727
- /** UUID for the run. */
728
- runId: string;
729
- /** Logical experiment grouping (a treatment vs a baseline within
730
- * the same sweep should share `experimentId`). */
731
- experimentId: string;
732
- /** Stable identifier for the candidate (variant) being run. The
733
- * promotion gate compares two `candidateId`s on matched items. */
734
- candidateId: string;
735
- /** RNG seed for the run. Always recorded — silent re-seeding is
736
- * the most common cause of non-reproducible numbers. */
737
- seed: number;
738
- /** Model identifier WITH snapshot version. */
739
- model: string;
740
- /** sha256 of the effective prompt (post-steering). */
741
- promptHash: string;
742
- /** sha256 of the effective config. */
743
- configHash: string;
744
- /** Git SHA the harness was run from. */
745
- commitSha: string;
746
- /** End-to-end wall-clock duration in milliseconds. */
747
- wallMs: number;
748
- /** Time spent queued before execution started, if known. */
749
- queueMs?: number;
750
- /** Total USD cost. Mandatory — runs without a cost number are
751
- * unbounded by definition and must not be admitted into the gate.
752
- * `0` is retained as the compatibility sentinel for an uncaptured amount;
753
- * inspect `costProvenance` before treating it as observed. */
754
- costUsd: number;
755
- /** Observed, model-priced estimate, or genuinely uncaptured USD amount.
756
- * Optional only so existing serialized RunRecords remain valid. */
757
- costProvenance?: RunCostProvenance;
758
- /** Token usage breakdown. */
759
- tokenUsage: RunTokenUsage;
760
- /** Judge-side metadata, if a judge was used. */
761
- judgeMetadata?: RunJudgeMetadata;
762
- /** Per-split scores + raw bag. */
763
- outcome: RunOutcome;
764
- /** Canonical, cross-agent failure class drawn from the shared
765
- * `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
766
- * "which failure dominates across the whole fleet" answerable in ONE
767
- * vocabulary — every agent classifies against the same enum. Producers
768
- * set it via the substrate classifier; leave unset only when the failure
769
- * genuinely can't be classified. */
770
- failureClass?: FailureClass;
771
- /** Free-form domain-specific failure detail, scoped UNDER `failureClass`
772
- * (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
773
- * The within-agent drill-down; `failureClass` is the cross-agent key. */
774
- failureMode?: string;
775
- /** Which split this run was drawn from. */
776
- splitTag: RunSplitTag;
777
- /**
778
- * Stable scenario identifier the run was scored against. Optional for
779
- * backwards compatibility, but **strongly recommended**: every primitive
780
- * that pairs runs by scenario (preferences, paired stats, BT tournament)
781
- * keys on this. The campaign artifact populates it canonically; legacy
782
- * runs without it fall back to inference from `outcome.raw.scenario_id`
783
- * or `experimentId`.
784
- */
785
- scenarioId?: string;
786
- /**
787
- * Canonical identity for the agent profile cell that produced this row:
788
- * profile artifact hash plus optional harness/model/prompt/reporting
789
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
790
- * longitudinal reports by the complete source profile, not by a loose
791
- * candidate label or opaque config hash.
792
- */
793
- agentProfile?: AgentProfileCell;
794
- }
795
-
796
- type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
797
- interface CodeAgentSessionMetrics {
798
- entries: number;
799
- userMessages: number;
800
- assistantMessages: number;
801
- reasoningItems: number;
802
- toolCalls: number;
803
- toolOutputs: number;
804
- toolErrors: number;
805
- patchAttempts: number;
806
- patchSuccesses: number;
807
- patchFailures: number;
808
- turnsStarted: number;
809
- turnsCompleted: number;
810
- turnsAborted: number;
811
- contextCompactions: number;
812
- prLinks: number;
813
- fileSnapshots: number;
814
- graphNodes: number;
815
- graphEdges: number;
816
- actionCandidates: number;
817
- verificationReports: number;
818
- completionDecisions: number;
819
- reliabilityRows: number;
820
- reliabilityLift: number;
821
- inputTokens: number;
822
- outputTokens: number;
823
- reasoningTokens: number;
824
- cachedTokens: number;
825
- cacheWriteTokens: number;
826
- observedCostUsd: number;
827
- observedCostCaptured?: boolean;
828
- wallMs: number;
829
- processScore: number;
830
- }
831
- interface CodeAgentSessionDiagnostic {
832
- source: CodeAgentSessionSource;
833
- sessionId: string;
834
- sourcePath?: string;
835
- entries: number;
836
- malformedLines: number;
837
- inferredScore: boolean;
838
- hasExplicitTerminalSignal: boolean;
839
- hasQualityLabel: boolean;
840
- hasTokenUsage: boolean;
841
- hasCost: boolean;
842
- costKind?: RunCostProvenance['kind'];
843
- warnings: string[];
844
- }
845
- interface CodeAgentSessionIntakeOptions {
846
- entries: unknown[];
847
- malformedLines?: number;
848
- sourcePath?: string;
849
- experimentId?: string;
850
- candidateId?: string;
851
- seed?: number;
852
- splitTag?: RunSplitTag;
853
- scenarioId?: string;
854
- model?: string;
855
- promptHash?: string;
856
- configHash?: string;
857
- commitSha?: string;
858
- score?: number;
859
- /** Explicit cost receipt. Use `uncaptured` when the source says dollars
860
- * were not captured; the adapter will not relabel its compatibility $0
861
- * sentinel as observed. When omitted, source-reported cost wins, then a
862
- * token-priced estimate, then uncaptured. */
863
- costProvenance?: RunCostProvenance;
864
- }
865
-
866
- interface BeliefOpeOptions extends OffPolicyOptions {
867
- minEffectiveSampleSize?: number;
868
- minEffectiveSampleRatio?: number;
869
- maxDiagnostics?: number;
870
- }
871
- interface BeliefOffPolicyTrajectoryReport {
872
- targetPolicyId: string;
873
- trajectories: OffPolicyTrajectory[];
874
- dropped: number;
875
- diagnostics: string[];
876
- }
877
- declare function embeddedBeliefOpeTargetPolicy(id?: string): BeliefOpeTargetPolicy;
878
- declare function beliefDecisionsToOffPolicyTrajectories(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: Pick<BeliefOpeOptions, 'maxDiagnostics'>): BeliefOffPolicyTrajectoryReport;
879
- declare function evaluateBeliefOffPolicy(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: BeliefOpeOptions): BeliefOpeReport;
880
-
881
- interface EvaluateBeliefSelectivePolicyOptions {
882
- utility?: BeliefUtilityOptions;
883
- minN?: number;
884
- minAccepted?: number;
885
- minUtilityDelta?: number;
886
- seed?: number;
887
- }
888
- declare function thresholdSelectivePolicy(options: {
889
- id?: string;
890
- confidenceThreshold: number;
891
- belowThresholdAction?: Exclude<BeliefPolicyAction, 'accept'>;
892
- }): BeliefSelectivePolicy;
893
- declare function evaluateBeliefSelectivePolicy(points: BeliefDecisionPoint[], policy: BeliefSelectivePolicy, options?: EvaluateBeliefSelectivePolicyOptions): BeliefSelectivePolicyMetrics;
894
-
895
- interface AnalyzeBeliefPolicyOpeOptions extends BeliefOpeOptions {
896
- targetPolicy?: BeliefOpeTargetPolicy;
897
- }
898
- interface AnalyzeBeliefPolicyOptions {
899
- points: BeliefDecisionPoint[];
900
- policy: BeliefSelectivePolicy;
901
- selective?: EvaluateBeliefSelectivePolicyOptions;
902
- calibration?: BeliefCalibrationOptions;
903
- ope?: AnalyzeBeliefPolicyOpeOptions;
904
- requireOpe?: boolean;
905
- }
906
- declare function analyzeBeliefPolicy(options: AnalyzeBeliefPolicyOptions): BeliefPolicyEvaluationReport;
907
-
908
- type CodeAgentBeliefDecisionTargetId = 'failure-recovery' | 'tool-selection' | 'graph-completion';
909
- interface ExtractCodeAgentBeliefDecisionPointsOptions {
910
- source: CodeAgentSessionSource;
911
- entries: unknown[];
912
- run: Pick<RunRecord, 'runId' | 'scenarioId' | 'outcome' | 'costUsd'>;
913
- sourcePath?: string;
914
- }
915
- interface BeliefDecisionInventoryBucket {
916
- id: string;
917
- kind?: BeliefDecisionKind;
918
- targetId?: CodeAgentBeliefDecisionTargetId;
919
- n: number;
920
- withOutcome: number;
921
- withConfidence: number;
922
- withCandidateActions: number;
923
- withBehaviorProb: number;
924
- withTargetProb: number;
925
- successRate: number | null;
926
- meanScore: number | null;
927
- meanConfidence: number | null;
928
- }
929
- interface BeliefDecisionInventoryReport {
930
- n: number;
931
- byKind: BeliefDecisionInventoryBucket[];
932
- byTarget: BeliefDecisionInventoryBucket[];
933
- diagnostics: string[];
934
- }
935
- interface BeliefDecisionTargetSelection {
936
- id: CodeAgentBeliefDecisionTargetId;
937
- label: string;
938
- points: BeliefDecisionPoint[];
939
- support: BeliefDecisionInventoryBucket;
940
- reasons: string[];
941
- }
942
- interface SelectBeliefDecisionTargetOptions {
943
- minN?: number;
944
- minOutcomeCoverage?: number;
945
- preferredTargets?: CodeAgentBeliefDecisionTargetId[];
946
- }
947
- interface AnalyzeBeliefDecisionCorpusOptions {
948
- points: BeliefDecisionPoint[];
949
- targetId?: CodeAgentBeliefDecisionTargetId;
950
- minN?: number;
951
- minOutcomeCoverage?: number;
952
- minAccepted?: number;
953
- confidenceThreshold?: number;
954
- policy?: BeliefSelectivePolicy;
955
- requireOpe?: boolean;
956
- policyOptions?: Partial<AnalyzeBeliefPolicyOptions>;
957
- }
958
- interface BeliefDecisionCorpusEvaluation {
959
- inventory: BeliefDecisionInventoryReport;
960
- target?: BeliefDecisionTargetSelection;
961
- policy?: BeliefSelectivePolicy;
962
- evaluation?: BeliefPolicyEvaluationReport;
963
- diagnostics: string[];
964
- }
965
- declare function extractCodeAgentBeliefDecisionPoints(options: ExtractCodeAgentBeliefDecisionPointsOptions): BeliefDecisionExtractionReport;
966
- declare function inventoryBeliefDecisionPoints(points: BeliefDecisionPoint[]): BeliefDecisionInventoryReport;
967
- declare function selectBeliefDecisionTarget(points: BeliefDecisionPoint[], options?: SelectBeliefDecisionTargetOptions): BeliefDecisionTargetSelection | null;
968
- declare function analyzeBeliefDecisionCorpus(options: AnalyzeBeliefDecisionCorpusOptions): BeliefDecisionCorpusEvaluation;
969
-
970
- type BeliefResearchClaimScope = 'selective' | 'counterfactual';
971
- type BeliefResearchEvidenceStatus = 'supported' | 'blocked';
972
- type BeliefResearchGateId = 'corpus' | 'selective' | 'calibration' | 'ope';
973
- interface BeliefResearchEvidenceGate {
974
- id: BeliefResearchGateId;
975
- status: BeliefResearchEvidenceStatus;
976
- blockers: string[];
977
- caveats: string[];
978
- }
979
- interface BeliefDecisionResearchEvidencePacket {
980
- claimScope: BeliefResearchClaimScope;
981
- status: BeliefResearchEvidenceStatus;
982
- analysis: BeliefDecisionCorpusEvaluation;
983
- gates: BeliefResearchEvidenceGate[];
984
- blockers: string[];
985
- caveats: string[];
986
- }
987
- interface BuildBeliefDecisionResearchEvidencePacketOptions extends AnalyzeBeliefDecisionCorpusOptions {
988
- claimScope?: BeliefResearchClaimScope;
989
- }
990
- declare function buildBeliefDecisionResearchEvidencePacket(options: BuildBeliefDecisionResearchEvidencePacketOptions): BeliefDecisionResearchEvidencePacket;
991
-
992
- interface CodeAgentBeliefSession extends CodeAgentSessionIntakeOptions {
993
- source: CodeAgentSessionSource;
994
- }
995
- interface BuildCodeAgentBeliefEvidenceCorpusOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
996
- sessions: CodeAgentBeliefSession[];
997
- }
998
- interface CodeAgentBeliefEvidenceCorpus {
999
- runs: RunRecord[];
1000
- metrics: CodeAgentSessionMetrics[];
1001
- intakeDiagnostics: CodeAgentSessionDiagnostic[];
1002
- extractionDiagnostics: BeliefDecisionExtractionDiagnostic[];
1003
- decisions: BeliefDecisionPoint[];
1004
- inventory: BeliefDecisionInventoryReport;
1005
- evidence: BeliefDecisionResearchEvidencePacket;
1006
- }
1007
- declare function buildCodeAgentBeliefEvidenceCorpus(options: BuildCodeAgentBeliefEvidenceCorpusOptions): CodeAgentBeliefEvidenceCorpus;
1008
-
1009
- interface ExtractBeliefDecisionPointsOptions {
1010
- runIds?: string[];
1011
- }
1012
- declare function extractBeliefDecisionPoints(store: TraceStore, options?: ExtractBeliefDecisionPointsOptions): Promise<BeliefDecisionExtractionReport>;
1013
-
1014
- interface BeliefShadowProbeInput {
1015
- probeId: string;
1016
- decisionId: string;
1017
- runId: string;
1018
- scenarioId?: string;
1019
- stepIndex: number;
1020
- decisionKind: BeliefDecisionKind;
1021
- candidateActions: string[];
1022
- observedAction?: string;
1023
- evidence: BeliefShadowProbeEvidenceRef[];
1024
- context?: string;
1025
- metadata?: Record<string, unknown>;
1026
- }
1027
- interface BeliefShadowProbeEvidenceRef {
1028
- id: string;
1029
- source: string;
1030
- detail?: string;
1031
- quality?: BeliefEvidenceQuality;
1032
- }
1033
- interface BeliefShadowProbeResponse {
1034
- predictedAction: string;
1035
- confidence: number;
1036
- beliefSummary?: string;
1037
- uncertainty?: string[];
1038
- evidenceRefs?: string[];
1039
- wouldChangeMindIf?: string[];
1040
- targetProb?: number;
1041
- qHat?: number | null;
1042
- metadata?: Record<string, unknown>;
1043
- }
1044
- interface BeliefShadowProbeRecord extends BeliefShadowProbeResponse {
1045
- probeId: string;
1046
- decisionId: string;
1047
- runId: string;
1048
- scenarioId?: string;
1049
- stepIndex: number;
1050
- decisionKind: BeliefDecisionKind;
1051
- candidateActions: string[];
1052
- observedAction: string;
1053
- agreesWithObservedAction: boolean;
1054
- outcome?: BeliefDecisionOutcome;
1055
- }
1056
- interface BeliefShadowProbeDiagnostic {
1057
- decisionId: string;
1058
- severity: 'warning' | 'error';
1059
- reason: string;
1060
- }
1061
- interface BeliefShadowProbeSummary {
1062
- attempted: number;
1063
- completed: number;
1064
- dropped: number;
1065
- withOutcome: number;
1066
- withTargetProb: number;
1067
- meanConfidence: number | null;
1068
- observedAgreementRate: number | null;
1069
- }
1070
- interface BeliefShadowProbeRun {
1071
- probeId: string;
1072
- records: BeliefShadowProbeRecord[];
1073
- diagnostics: BeliefShadowProbeDiagnostic[];
1074
- summary: BeliefShadowProbeSummary;
1075
- }
1076
- interface RunBeliefShadowProbeOptions {
1077
- probeId: string;
1078
- points: BeliefDecisionPoint[];
1079
- probe: (input: BeliefShadowProbeInput) => BeliefShadowProbeResponse | Promise<BeliefShadowProbeResponse>;
1080
- contextOf?: (point: BeliefDecisionPoint) => string | undefined | Promise<string | undefined>;
1081
- metadataOf?: (point: BeliefDecisionPoint) => Record<string, unknown> | undefined | Promise<Record<string, unknown> | undefined>;
1082
- includeObservedAction?: boolean;
1083
- includeEvidenceDetail?: boolean;
1084
- includeOutcomeInRecord?: boolean;
1085
- requireCandidateActions?: boolean;
1086
- allowOutOfSetActions?: boolean;
1087
- concurrency?: number;
1088
- maxContextChars?: number;
1089
- }
1090
- declare function runBeliefShadowProbe(options: RunBeliefShadowProbeOptions): Promise<BeliefShadowProbeRun>;
1091
- declare function formatBeliefShadowProbePrompt(input: BeliefShadowProbeInput): string;
1092
-
1093
- interface RuntimeBeliefDecisionEvidenceRef {
1094
- source: string;
1095
- id: string;
1096
- detail?: string;
1097
- quality?: BeliefEvidenceQuality;
1098
- metadata?: Record<string, unknown>;
1099
- }
1100
- interface RuntimeBeliefDecisionPoint {
1101
- id: string;
1102
- runId: string;
1103
- scenarioId?: string;
1104
- stepIndex: number;
1105
- kind: string;
1106
- candidateActions?: string[];
1107
- context?: string;
1108
- evidence?: RuntimeBeliefDecisionEvidenceRef[];
1109
- metadata?: Record<string, unknown>;
1110
- }
1111
- interface RuntimeBeliefHookEvent {
1112
- id: string;
1113
- runId: string;
1114
- scenarioId?: string;
1115
- target: string;
1116
- phase: string;
1117
- timestamp: number;
1118
- stepIndex?: number;
1119
- parentId?: string;
1120
- payload?: unknown;
1121
- metadata?: Record<string, unknown>;
1122
- }
1123
- interface RuntimeBeliefHookContext {
1124
- signal?: AbortSignal;
1125
- }
1126
- interface RuntimeBeliefHooks {
1127
- onEvent?: (event: RuntimeBeliefHookEvent, context: RuntimeBeliefHookContext) => void | Promise<void>;
1128
- onDecisionPoint?: (point: RuntimeBeliefDecisionPoint, context: RuntimeBeliefHookContext) => void | Promise<void>;
1129
- }
1130
- interface RuntimeBeliefConversionDiagnostic {
1131
- decisionId: string;
1132
- severity: 'warning' | 'error';
1133
- reason: string;
1134
- }
1135
- interface RuntimeBeliefShadowProbeInputOptions {
1136
- probeId: string;
1137
- decisionKind?: BeliefDecisionKind;
1138
- includeEvidenceDetail?: boolean;
1139
- includeLifecycleEvidence?: boolean;
1140
- lifecycleEvents?: RuntimeBeliefHookEvent[];
1141
- maxContextChars?: number;
1142
- }
1143
- interface RuntimeBeliefDecisionPointOptions {
1144
- chosenAction?: string;
1145
- decisionKind?: BeliefDecisionKind;
1146
- confidence?: number;
1147
- behaviorProb?: number;
1148
- targetProb?: number;
1149
- qHat?: number | null;
1150
- costUsd?: number;
1151
- outcome?: BeliefDecisionOutcome;
1152
- metadata?: Record<string, unknown>;
1153
- includeLifecycleEvidence?: boolean;
1154
- lifecycleEvents?: RuntimeBeliefHookEvent[];
1155
- }
1156
- interface RuntimeBeliefShadowProbeInputReport {
1157
- input?: BeliefShadowProbeInput;
1158
- diagnostics: RuntimeBeliefConversionDiagnostic[];
1159
- }
1160
- interface RuntimeBeliefDecisionPointReport {
1161
- point?: BeliefDecisionPoint;
1162
- diagnostics: RuntimeBeliefConversionDiagnostic[];
1163
- }
1164
- interface BeliefRuntimeHookCollector {
1165
- hooks: RuntimeBeliefHooks;
1166
- decisions: RuntimeBeliefDecisionPoint[];
1167
- events: RuntimeBeliefHookEvent[];
1168
- toShadowProbeInputs(options?: Partial<RuntimeBeliefShadowProbeInputOptions>): {
1169
- inputs: BeliefShadowProbeInput[];
1170
- diagnostics: RuntimeBeliefConversionDiagnostic[];
1171
- };
1172
- clear(): void;
1173
- }
1174
- declare function runtimeDecisionPointToBeliefShadowProbeInput(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefShadowProbeInputOptions): RuntimeBeliefShadowProbeInputReport;
1175
- declare function runtimeDecisionPointToBeliefDecisionPoint(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefDecisionPointOptions): RuntimeBeliefDecisionPointReport;
1176
- declare function createBeliefRuntimeHookCollector(defaults: RuntimeBeliefShadowProbeInputOptions): BeliefRuntimeHookCollector;
1177
-
1178
- interface RuntimeBeliefPhase0RunRecord {
1179
- runId: string;
1180
- scenarioId?: string;
1181
- splitTag: RunSplitTag;
1182
- }
1183
- interface RuntimeBeliefDecisionLabel {
1184
- decisionId: string;
1185
- chosenAction: string;
1186
- outcome: BeliefDecisionOutcome;
1187
- confidence?: number;
1188
- behaviorProb?: number;
1189
- targetProb?: number;
1190
- qHat?: number | null;
1191
- costUsd?: number;
1192
- splitTag?: RunSplitTag;
1193
- metadata?: Record<string, unknown>;
1194
- }
1195
- interface BuildRuntimeBeliefPhase0MeasurementOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
1196
- runs: RuntimeBeliefPhase0RunRecord[];
1197
- decisions: RuntimeBeliefDecisionPoint[];
1198
- events?: RuntimeBeliefHookEvent[];
1199
- labels: RuntimeBeliefDecisionLabel[];
1200
- baselinePolicyId?: string;
1201
- }
1202
- interface RuntimeBeliefPhase0MeasurementSummary {
1203
- runCount: number;
1204
- producerDecisionCount: number;
1205
- lifecycleEventCount: number;
1206
- labelCount: number;
1207
- completedPointCount: number;
1208
- runJoinRate: number;
1209
- labelJoinRate: number;
1210
- missingRunRecordCount: number;
1211
- missingLabelCount: number;
1212
- withEvidence: number;
1213
- withOutcome: number;
1214
- withSplit: number;
1215
- withBehaviorProb: number;
1216
- withTargetProb: number;
1217
- baselinePolicyId: string;
1218
- packetStatus: BeliefDecisionResearchEvidencePacket['status'];
1219
- claimScope: BeliefDecisionResearchEvidencePacket['claimScope'];
1220
- }
1221
- interface RuntimeBeliefPhase0Measurement {
1222
- points: BeliefDecisionPoint[];
1223
- packet: BeliefDecisionResearchEvidencePacket;
1224
- summary: RuntimeBeliefPhase0MeasurementSummary;
1225
- diagnostics: string[];
1226
- }
1227
- declare function buildRuntimeBeliefPhase0Measurement(options: BuildRuntimeBeliefPhase0MeasurementOptions): RuntimeBeliefPhase0Measurement;
1228
-
1229
- interface RuntimeTrajectoryHookEvent {
1230
- id: string;
1231
- runId: string;
1232
- scenarioId?: string;
1233
- target: string;
1234
- phase: string;
1235
- timestamp: number;
1236
- stepIndex?: number;
1237
- parentId?: string;
1238
- payload?: unknown;
1239
- metadata?: Record<string, unknown>;
1240
- }
1241
- interface RuntimeTrajectoryRecord {
1242
- id?: string;
1243
- scenarioId?: string;
1244
- splitTag?: RunSplitTag;
1245
- runtimeEvents?: unknown;
1246
- [key: string]: unknown;
1247
- }
1248
- interface RuntimeTrajectoryRunRecord {
1249
- runId: string;
1250
- scenarioId?: string;
1251
- splitTag: RunSplitTag;
1252
- }
1253
- interface RuntimeTrajectoryEvidenceSummary {
1254
- recordCount: number;
1255
- recordWithRuntimeEventsCount: number;
1256
- runtimeRunCount: number;
1257
- lifecycleEventCount: number;
1258
- defaultedSplitCount: number;
1259
- }
1260
- interface RuntimeTrajectoryEvidenceProjection {
1261
- runs: RuntimeTrajectoryRunRecord[];
1262
- events: RuntimeTrajectoryHookEvent[];
1263
- summary: RuntimeTrajectoryEvidenceSummary;
1264
- diagnostics: string[];
1265
- }
1266
- interface ProjectRuntimeTrajectoryEvidenceOptions<TRecord extends RuntimeTrajectoryRecord = RuntimeTrajectoryRecord> {
1267
- records: TRecord[];
1268
- defaultSplitTag?: RunSplitTag;
1269
- recordIdOf?: (record: TRecord, index: number) => string | undefined;
1270
- scenarioIdOf?: (record: TRecord, index: number) => string | undefined;
1271
- }
1272
-
1273
- type RuntimeBenchmarkTrajectoryRecord = RuntimeTrajectoryRecord & {
1274
- benchmark?: unknown;
1275
- condition?: unknown;
1276
- instanceId?: unknown;
1277
- runtimeDecisionPoints?: unknown;
1278
- };
1279
- interface BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions extends Omit<BuildRuntimeBeliefPhase0MeasurementOptions, 'runs' | 'events' | 'decisions' | 'labels'> {
1280
- records: RuntimeBenchmarkTrajectoryRecord[];
1281
- decisions?: RuntimeBeliefDecisionPoint[];
1282
- defaultSplitTag?: ProjectRuntimeTrajectoryEvidenceOptions['defaultSplitTag'];
1283
- labels?: RuntimeBeliefDecisionLabel[];
1284
- }
1285
- interface RuntimeBenchmarkBeliefPhase0Summary {
1286
- decisionCount: number;
1287
- labelCount: number;
1288
- }
1289
- interface RuntimeBenchmarkBeliefPhase0Measurement {
1290
- runs: RuntimeBeliefPhase0RunRecord[];
1291
- events: RuntimeBeliefHookEvent[];
1292
- decisions: RuntimeBeliefDecisionPoint[];
1293
- labels: RuntimeBeliefDecisionLabel[];
1294
- trajectory: RuntimeTrajectoryEvidenceProjection;
1295
- measurement: RuntimeBeliefPhase0Measurement;
1296
- summary: RuntimeBenchmarkBeliefPhase0Summary;
1297
- diagnostics: string[];
1298
- }
1299
- declare function buildRuntimeBenchmarkBeliefPhase0Measurement(options: BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions): RuntimeBenchmarkBeliefPhase0Measurement;
1300
-
1301
- export { type AnalyzeBeliefDecisionCorpusOptions, type AnalyzeBeliefPolicyOpeOptions, type AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, type BeliefCalibrationOptions, type BeliefCalibrationRegion, type BeliefCalibrationStatus, type BeliefDecisionCorpusEvaluation, type BeliefDecisionExtractionDiagnostic, type BeliefDecisionExtractionReport, type BeliefDecisionInventoryBucket, type BeliefDecisionInventoryReport, type BeliefDecisionKind, type BeliefDecisionOutcome, type BeliefDecisionPoint, type BeliefDecisionReason, type BeliefDecisionReasonCode, type BeliefDecisionResearchEvidencePacket, type BeliefDecisionTargetSelection, type BeliefEvaluationCriterionId, type BeliefEvaluationStatus, type BeliefEvidenceQuality, type BeliefEvidenceRef, type BeliefEvidenceSource, type BeliefOffPolicyTrajectoryReport, type BeliefOpeOptions, type BeliefOpeReport, type BeliefOpeStatus, type BeliefOpeSupportDiagnostics, type BeliefOpeTargetPolicy, type BeliefPolicyAction, type BeliefPolicyDecision, type BeliefPolicyEvaluationReport, type BeliefResearchClaimScope, type BeliefResearchEvidenceGate, type BeliefResearchEvidenceStatus, type BeliefResearchGateId, type BeliefRuntimeHookCollector, type BeliefSelectivePolicy, type BeliefSelectivePolicyMetrics, type BeliefShadowProbeDiagnostic, type BeliefShadowProbeEvidenceRef, type BeliefShadowProbeInput, type BeliefShadowProbeRecord, type BeliefShadowProbeResponse, type BeliefShadowProbeRun, type BeliefShadowProbeSummary, type BeliefUtilityOptions, type BuildBeliefDecisionResearchEvidencePacketOptions, type BuildCodeAgentBeliefEvidenceCorpusOptions, type BuildRuntimeBeliefPhase0MeasurementOptions, type BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions, type CodeAgentBeliefDecisionTargetId, type CodeAgentBeliefEvidenceCorpus, type CodeAgentBeliefSession, type EvaluateBeliefSelectivePolicyOptions, type ExtractBeliefDecisionPointsOptions, type ExtractCodeAgentBeliefDecisionPointsOptions, type RunBeliefShadowProbeOptions, type RuntimeBeliefConversionDiagnostic, type RuntimeBeliefDecisionEvidenceRef, type RuntimeBeliefDecisionLabel, type RuntimeBeliefDecisionPoint, type RuntimeBeliefDecisionPointOptions, type RuntimeBeliefDecisionPointReport, type RuntimeBeliefHookContext, type RuntimeBeliefHookEvent, type RuntimeBeliefHooks, type RuntimeBeliefPhase0Measurement, type RuntimeBeliefPhase0MeasurementSummary, type RuntimeBeliefPhase0RunRecord, type RuntimeBeliefShadowProbeInputOptions, type RuntimeBeliefShadowProbeInputReport, type RuntimeBenchmarkBeliefPhase0Measurement, type RuntimeBenchmarkBeliefPhase0Summary, type SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };