headlesscode 1.2.0 → 1.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,73 @@
1
+ import type {
2
+ FrozenPredictorArtifact,
3
+ InterventionDecision,
4
+ IterationInterventionTemplateId,
5
+ PilotArm,
6
+ PilotBoundary,
7
+ Prediction,
8
+ } from "./types.js"
9
+ import { predict, predictedIntervention } from "./predictor.js"
10
+
11
+ export const SPECIFIC_CORRECTION =
12
+ "Check the latest tool result. Re-read the affected configuration if necessary, identify the failed precondition, and change the next action before retrying."
13
+ export const GENERIC_REVIEW = "Pause and review your approach before continuing."
14
+ export const TEMPLATE_TEXT: Readonly<Record<IterationInterventionTemplateId, string>> = {
15
+ "specific-correction": SPECIFIC_CORRECTION,
16
+ "generic-review": GENERIC_REVIEW,
17
+ }
18
+
19
+ export interface ControllerState {
20
+ readonly arm: PilotArm
21
+ readonly disarmed: boolean
22
+ readonly appliedInterventionIds: ReadonlySet<string>
23
+ }
24
+
25
+ export function initialControllerState(arm: PilotArm): ControllerState {
26
+ return { arm, disarmed: false, appliedInterventionIds: new Set() }
27
+ }
28
+
29
+ export function decide(
30
+ artifact: FrozenPredictorArtifact | undefined,
31
+ boundary: PilotBoundary,
32
+ state: ControllerState,
33
+ prediction?: Prediction,
34
+ ): { prediction?: Prediction; decision?: InterventionDecision; state: ControllerState } {
35
+ if (state.disarmed || state.arm === "B") return { prediction, state }
36
+ const nextPrediction = prediction ?? (artifact === undefined ? undefined : predict(artifact, boundary))
37
+ if (nextPrediction === undefined || artifact === undefined || !predictedIntervention(artifact, nextPrediction)) {
38
+ return { prediction: nextPrediction, state }
39
+ }
40
+ const templateId: IterationInterventionTemplateId = state.arm === "G" ? "generic-review" : "specific-correction"
41
+ const interventionId = `${boundary.attemptId}:${boundary.sequence}:${templateId}`
42
+ if (state.appliedInterventionIds.has(interventionId)) return { prediction: nextPrediction, state }
43
+ const applied = new Set(state.appliedInterventionIds)
44
+ applied.add(interventionId)
45
+ return {
46
+ prediction: nextPrediction,
47
+ decision: { interventionId, templateId, reason: "threshold" },
48
+ state: { ...state, disarmed: true, appliedInterventionIds: applied },
49
+ }
50
+ }
51
+
52
+ export function randomDecision(
53
+ boundary: PilotBoundary,
54
+ state: ControllerState,
55
+ probability: number,
56
+ random: () => number,
57
+ ): { decision?: InterventionDecision; state: ControllerState } {
58
+ if (state.disarmed || state.arm !== "R" || probability <= 0 || random() >= probability) return { state }
59
+ const interventionId = `${boundary.attemptId}:${boundary.sequence}:specific-correction`
60
+ if (state.appliedInterventionIds.has(interventionId)) return { state }
61
+ const applied = new Set(state.appliedInterventionIds)
62
+ applied.add(interventionId)
63
+ return {
64
+ decision: { interventionId, templateId: "specific-correction", reason: "random-schedule" },
65
+ state: { ...state, disarmed: true, appliedInterventionIds: applied },
66
+ }
67
+ }
68
+
69
+ export function applyFixedTemplate(messages: string[], decision: InterventionDecision): void {
70
+ const text = TEMPLATE_TEXT[decision.templateId]
71
+ if (typeof text !== "string") throw new Error("unknown intervention template")
72
+ messages.push(text)
73
+ }
@@ -0,0 +1,103 @@
1
+ import { createHash } from "node:crypto"
2
+ import type { MonitorFeatureName, MonitorFeatureVector, PilotBoundary, PilotToolCall, PilotToolResult } from "./types.js"
3
+ import { MONITOR_FEATURE_ORDER } from "./types.js"
4
+
5
+ export function stableJson(value: unknown): string {
6
+ if (value === null || typeof value !== "object") return JSON.stringify(value)
7
+ if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`
8
+ const record = value as Record<string, unknown>
9
+ return `{${Object.keys(record)
10
+ .sort()
11
+ .map((key) => `${JSON.stringify(key)}:${stableJson(record[key])}`)
12
+ .join(",")}}`
13
+ }
14
+
15
+ export function sha256(value: unknown): string {
16
+ return createHash("sha256").update(typeof value === "string" ? value : stableJson(value)).digest("hex")
17
+ }
18
+
19
+ function isReadOnlyCall(call: PilotToolCall): boolean {
20
+ return call.name === "read_file" || call.name === "list_files" || call.name === "execute_command"
21
+ }
22
+
23
+ function isWriteCall(call: PilotToolCall): boolean {
24
+ return call.name === "apply_diff"
25
+ }
26
+
27
+ function callSignature(call: PilotToolCall): string {
28
+ return `${call.name}:${stableJson(call.args)}`
29
+ }
30
+
31
+ function resultSignature(result: PilotToolResult): string {
32
+ return `${result.name}:${result.isError}:${result.resultHash}`
33
+ }
34
+
35
+ function boundedFraction(numerator: number, denominator: number): number {
36
+ return denominator > 0 ? numerator / denominator : 0
37
+ }
38
+
39
+ /** Extract only current and recent action/result state. Oracle fields never enter here. */
40
+ export function extractMonitorFeatures(boundary: PilotBoundary): MonitorFeatureVector {
41
+ const turns = [...boundary.recentTurns].slice(-4)
42
+ const calls = turns.flatMap((turn) => turn.calls)
43
+ const results = turns.flatMap((turn) => turn.results)
44
+ const lastCall = boundary.calls.at(-1)
45
+ let identicalCallStreak = boundary.deterministicGuardrails.identicalCallStreak
46
+ if (lastCall !== undefined) {
47
+ identicalCallStreak = 0
48
+ for (const call of [...calls].reverse()) {
49
+ if (callSignature(call) !== callSignature(lastCall)) break
50
+ identicalCallStreak++
51
+ }
52
+ }
53
+ let sameToolErrorStreak = boundary.deterministicGuardrails.sameToolErrorStreak
54
+ const lastError = [...results].reverse().find((result) => result.isError)
55
+ if (lastError !== undefined) {
56
+ sameToolErrorStreak = 0
57
+ for (const result of [...results].reverse()) {
58
+ if (!result.isError || result.name !== lastError.name) break
59
+ sameToolErrorStreak++
60
+ }
61
+ }
62
+ const distinctCalls = new Set(calls.map(callSignature)).size
63
+ const errorCount = results.filter((result) => result.isError).length
64
+ const repeatedResults = results.length - new Set(results.map(resultSignature)).size
65
+ let turnsSinceSuccessfulVirtualWrite = 0
66
+ for (const turn of [...turns].reverse()) {
67
+ if (turn.successfulVirtualWrite) break
68
+ turnsSinceSuccessfulVirtualWrite++
69
+ }
70
+ const values: Record<MonitorFeatureName, number> = {
71
+ identicalCallStreak,
72
+ sameToolErrorStreak,
73
+ readOnlyStreak: boundary.deterministicGuardrails.readOnlyStreak,
74
+ uniqueCallFraction4: boundedFraction(distinctCalls, calls.length),
75
+ errorFraction4: boundedFraction(errorCount, results.length),
76
+ turnsSinceSuccessfulVirtualWrite,
77
+ repeatedResultFraction4: boundedFraction(repeatedResults, results.length),
78
+ iterationBudgetFraction: boundary.maxIterations > 0 ? boundary.iteration / boundary.maxIterations : 1,
79
+ }
80
+ return values
81
+ }
82
+
83
+ export function featureVector(features: MonitorFeatureVector): number[] {
84
+ return MONITOR_FEATURE_ORDER.map((name) => features[name])
85
+ }
86
+
87
+ export function featureHash(features: MonitorFeatureVector): string {
88
+ return sha256(featureVector(features))
89
+ }
90
+
91
+ export function assertFiniteFeatures(features: MonitorFeatureVector): void {
92
+ for (const name of MONITOR_FEATURE_ORDER) {
93
+ if (!Number.isFinite(features[name])) throw new Error(`non-finite feature: ${name}`)
94
+ }
95
+ }
96
+
97
+ export function currentTurnIsReadOnly(boundary: PilotBoundary): boolean {
98
+ return boundary.calls.length > 0 && boundary.calls.every(isReadOnlyCall)
99
+ }
100
+
101
+ export function currentTurnHasWrite(boundary: PilotBoundary): boolean {
102
+ return boundary.calls.some(isWriteCall) && boundary.results.some((result) => result.name === "apply_diff" && !result.isError)
103
+ }
@@ -0,0 +1,4 @@
1
+ export * from "./types.js"
2
+ export * from "./features.js"
3
+ export * from "./predictor.js"
4
+ export * from "./controller.js"
@@ -0,0 +1,246 @@
1
+ import type {
2
+ Calibration,
3
+ FrozenPredictorArtifact,
4
+ MonitorFeatureName,
5
+ MonitorFeatureVector,
6
+ PilotBoundary,
7
+ Prediction,
8
+ Standardization,
9
+ } from "./types.js"
10
+ import { MONITOR_FEATURE_ORDER } from "./types.js"
11
+ import { assertFiniteFeatures, extractMonitorFeatures, featureHash, sha256 } from "./features.js"
12
+
13
+ export interface LabeledWindow {
14
+ readonly features: MonitorFeatureVector
15
+ readonly label: 0 | 1
16
+ readonly trajectoryId: string
17
+ readonly boundarySequence?: number
18
+ }
19
+
20
+ export interface CalibrationPoint {
21
+ readonly score: number
22
+ readonly label: 0 | 1
23
+ }
24
+
25
+ export interface FitOptions {
26
+ readonly modelDigest: string
27
+ readonly templatesHash: string
28
+ readonly l2?: 1
29
+ }
30
+
31
+ function sigmoid(value: number): number {
32
+ if (value >= 0) {
33
+ const z = Math.exp(-value)
34
+ return 1 / (1 + z)
35
+ }
36
+ const z = Math.exp(value)
37
+ return z / (1 + z)
38
+ }
39
+
40
+ function emptyFeatureRecord(value: number): Record<MonitorFeatureName, number> {
41
+ return Object.fromEntries(MONITOR_FEATURE_ORDER.map((name) => [name, value])) as Record<MonitorFeatureName, number>
42
+ }
43
+
44
+ function standardize(rows: readonly LabeledWindow[]): Standardization {
45
+ const mean = emptyFeatureRecord(0)
46
+ const scale = emptyFeatureRecord(1)
47
+ for (const row of rows) {
48
+ for (const name of MONITOR_FEATURE_ORDER) mean[name] += row.features[name]
49
+ }
50
+ for (const name of MONITOR_FEATURE_ORDER) mean[name] /= Math.max(1, rows.length)
51
+ for (const row of rows) {
52
+ for (const name of MONITOR_FEATURE_ORDER) {
53
+ const delta = row.features[name] - mean[name]
54
+ scale[name] += delta * delta
55
+ }
56
+ }
57
+ for (const name of MONITOR_FEATURE_ORDER) scale[name] = Math.sqrt((scale[name] - 1) / Math.max(1, rows.length)) || 1
58
+ return { mean, scale }
59
+ }
60
+
61
+ function standardizedVector(features: MonitorFeatureVector, normalization: Standardization): number[] {
62
+ assertFiniteFeatures(features)
63
+ return MONITOR_FEATURE_ORDER.map((name) => (features[name] - normalization.mean[name]) / normalization.scale[name])
64
+ }
65
+
66
+ function fitLogistic(rows: readonly LabeledWindow[], normalization: Standardization): {
67
+ coefficients: Record<MonitorFeatureName, number>
68
+ intercept: number
69
+ } {
70
+ const coefficients = emptyFeatureRecord(0)
71
+ let intercept = 0
72
+ // Fixed optimizer settings are part of the frozen baseline, not tunable data.
73
+ const learningRate = 0.08
74
+ for (let step = 0; step < 2_000; step++) {
75
+ const gradient = emptyFeatureRecord(0)
76
+ let interceptGradient = 0
77
+ for (const row of rows) {
78
+ const x = standardizedVector(row.features, normalization)
79
+ let logit = intercept
80
+ for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) logit += coefficients[MONITOR_FEATURE_ORDER[i]] * x[i]
81
+ const error = sigmoid(logit) - row.label
82
+ interceptGradient += error
83
+ for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) gradient[MONITOR_FEATURE_ORDER[i]] += error * x[i]
84
+ }
85
+ const denominator = Math.max(1, rows.length)
86
+ intercept -= learningRate * (interceptGradient / denominator)
87
+ for (const name of MONITOR_FEATURE_ORDER) {
88
+ // Fixed L2 coefficient 1.0; intercept is not regularized.
89
+ coefficients[name] -= learningRate * (gradient[name] / denominator + coefficients[name] / denominator)
90
+ }
91
+ }
92
+ return { coefficients, intercept }
93
+ }
94
+
95
+ function rawScore(artifact: Omit<FrozenPredictorArtifact, "artifactHash">, features: MonitorFeatureVector): number {
96
+ const x = standardizedVector(features, artifact.standardization)
97
+ let logit = artifact.intercept
98
+ for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) logit += artifact.coefficients[MONITOR_FEATURE_ORDER[i]] * x[i]
99
+ return sigmoid(logit)
100
+ }
101
+
102
+ function fitCalibration(points: readonly CalibrationPoint[]): Calibration {
103
+ let intercept = 0
104
+ let slope = 1
105
+ for (let step = 0; step < 1_000; step++) {
106
+ let gi = 0
107
+ let gs = 0
108
+ for (const point of points) {
109
+ const logit = Math.log(Math.min(1 - 1e-6, Math.max(1e-6, point.score)) / Math.max(1e-6, 1 - point.score))
110
+ const probability = sigmoid(intercept + slope * logit)
111
+ const error = probability - point.label
112
+ gi += error
113
+ gs += error * logit
114
+ }
115
+ const denominator = Math.max(1, points.length)
116
+ intercept -= 0.05 * gi / denominator
117
+ slope -= 0.05 * gs / denominator
118
+ }
119
+ return { intercept, slope }
120
+ }
121
+
122
+ function calibratedScore(artifact: Omit<FrozenPredictorArtifact, "artifactHash">, features: MonitorFeatureVector): number {
123
+ const score = rawScore(artifact, features)
124
+ const logit = Math.log(Math.min(1 - 1e-6, Math.max(1e-6, score)) / Math.max(1e-6, 1 - score))
125
+ return sigmoid(artifact.calibration.intercept + artifact.calibration.slope * logit)
126
+ }
127
+
128
+ export function chooseThresholdFromScores(rows: readonly CalibrationPoint[]): number | undefined {
129
+ for (let step = 10; step <= 19; step++) {
130
+ const threshold = step / 20
131
+ const selected = rows.filter((row) => row.score >= threshold)
132
+ const truePositive = selected.filter((row) => row.label === 1).length
133
+ const falsePositive = selected.filter((row) => row.label === 0).length
134
+ const positives = rows.filter((row) => row.label === 1).length
135
+ const negatives = rows.filter((row) => row.label === 0).length
136
+ const precision = selected.length === 0 ? 1 : truePositive / selected.length
137
+ const falsePositiveRate = negatives === 0 ? 0 : falsePositive / negatives
138
+ if (precision >= 0.8 && falsePositiveRate <= 0.1 && positives >= 10) return threshold
139
+ }
140
+ return undefined
141
+ }
142
+
143
+ export function fitFrozenPredictor(
144
+ rows: readonly LabeledWindow[],
145
+ calibrationRows: readonly LabeledWindow[],
146
+ options: FitOptions,
147
+ ): FrozenPredictorArtifact {
148
+ if (rows.length === 0) throw new Error("cannot fit predictor without training rows")
149
+ const standardization = standardize(rows)
150
+ const model = fitLogistic(rows, standardization)
151
+ const provisional: Omit<FrozenPredictorArtifact, "artifactHash"> = {
152
+ schemaVersion: 1,
153
+ featureOrder: MONITOR_FEATURE_ORDER,
154
+ coefficients: model.coefficients,
155
+ intercept: model.intercept,
156
+ standardization,
157
+ calibration: fitCalibration(
158
+ calibrationRows.map((row) => ({
159
+ score: rawScore({
160
+ schemaVersion: 1,
161
+ featureOrder: MONITOR_FEATURE_ORDER,
162
+ coefficients: model.coefficients,
163
+ intercept: model.intercept,
164
+ standardization,
165
+ calibration: { intercept: 0, slope: 1 },
166
+ threshold: 0.5,
167
+ l2: 1,
168
+ modelDigest: options.modelDigest,
169
+ templatesHash: options.templatesHash,
170
+ }, row.features),
171
+ label: row.label,
172
+ })),
173
+ ),
174
+ threshold: 0.5,
175
+ l2: 1,
176
+ modelDigest: options.modelDigest,
177
+ templatesHash: options.templatesHash,
178
+ }
179
+ const calibrationScores = calibrationRows.map((row) => ({
180
+ score: calibratedScore(provisional, row.features),
181
+ label: row.label,
182
+ }))
183
+ const threshold = chooseThresholdFromScores(calibrationScores)
184
+ if (threshold === undefined) throw new Error("STOP: no qualifying calibration threshold")
185
+ const frozen = { ...provisional, threshold }
186
+ return { ...frozen, artifactHash: sha256(frozen) }
187
+ }
188
+
189
+ export function predict(artifact: FrozenPredictorArtifact, boundary: PilotBoundary): Prediction {
190
+ const started = process.hrtime.bigint()
191
+ try {
192
+ return predictFromFeatures(artifact, extractMonitorFeatures(boundary))
193
+ } catch (error) {
194
+ const elapsedNs = Number(process.hrtime.bigint() - started)
195
+ const features = emptyFeatureRecord(Number.NaN)
196
+ return {
197
+ probability: 0,
198
+ threshold: artifact.threshold,
199
+ features,
200
+ featureHash: sha256(features),
201
+ predictorHash: artifact.artifactHash,
202
+ latencyNs: elapsedNs,
203
+ abstentionReason: error instanceof Error && error.message.includes("finite")
204
+ ? "non-finite"
205
+ : error instanceof Error && error.message.includes("unknown model")
206
+ ? "unknown-model"
207
+ : "malformed",
208
+ }
209
+ }
210
+ }
211
+
212
+ export function predictFromFeatures(artifact: FrozenPredictorArtifact, features: MonitorFeatureVector): Prediction {
213
+ const started = process.hrtime.bigint()
214
+ try {
215
+ assertFiniteFeatures(features)
216
+ if (artifact.modelDigest.trim() === "") throw new Error("unknown model")
217
+ const provisional = artifact as Omit<FrozenPredictorArtifact, "artifactHash">
218
+ const probability = calibratedScore(provisional, features)
219
+ return {
220
+ probability,
221
+ threshold: artifact.threshold,
222
+ features,
223
+ featureHash: featureHash(features),
224
+ predictorHash: artifact.artifactHash,
225
+ latencyNs: Number(process.hrtime.bigint() - started),
226
+ }
227
+ } catch (error) {
228
+ return {
229
+ probability: 0,
230
+ threshold: artifact.threshold,
231
+ features,
232
+ featureHash: sha256(features),
233
+ predictorHash: artifact.artifactHash,
234
+ latencyNs: Number(process.hrtime.bigint() - started),
235
+ abstentionReason: error instanceof Error && error.message.includes("finite")
236
+ ? "non-finite"
237
+ : error instanceof Error && error.message.includes("unknown model")
238
+ ? "unknown-model"
239
+ : "malformed",
240
+ }
241
+ }
242
+ }
243
+
244
+ export function predictedIntervention(artifact: FrozenPredictorArtifact, prediction: Prediction): boolean {
245
+ return prediction.abstentionReason === undefined && prediction.probability >= artifact.threshold
246
+ }
@@ -0,0 +1,192 @@
1
+ import type { IterationInterventionTemplateId } from "../engine/types.js"
2
+
3
+ export { type IterationInterventionTemplateId }
4
+
5
+ export const MONITOR_FEATURE_ORDER = [
6
+ "identicalCallStreak",
7
+ "sameToolErrorStreak",
8
+ "readOnlyStreak",
9
+ "uniqueCallFraction4",
10
+ "errorFraction4",
11
+ "turnsSinceSuccessfulVirtualWrite",
12
+ "repeatedResultFraction4",
13
+ "iterationBudgetFraction",
14
+ ] as const
15
+
16
+ export type MonitorFeatureName = (typeof MONITOR_FEATURE_ORDER)[number]
17
+ export type MonitorFeatureVector = Record<MonitorFeatureName, number>
18
+
19
+ export type PilotArm = "B" | "M" | "R" | "G"
20
+ export type FixtureFamily = "wrong-key-path" | "revision-mismatch" | "failed-validation" | "legitimate-investigation"
21
+ export type FixtureSplit = "train" | "calibration" | "test"
22
+
23
+ export interface PilotFileSet {
24
+ readonly [relativePath: string]: string
25
+ }
26
+
27
+ export interface PilotFixture {
28
+ readonly id: string
29
+ readonly family: FixtureFamily
30
+ readonly split: FixtureSplit
31
+ readonly task: string
32
+ readonly initialFiles: PilotFileSet
33
+ readonly targetFiles: PilotFileSet
34
+ readonly unrelatedPaths: readonly string[]
35
+ readonly validator: (files: PilotFileSet) => { ok: boolean; message: string; milestone: number }
36
+ readonly oracle: (files: PilotFileSet) => { success: boolean; reason: string; milestone: number }
37
+ }
38
+
39
+ export interface PilotToolCall {
40
+ readonly id: string
41
+ readonly name: string
42
+ readonly args: Readonly<Record<string, unknown>>
43
+ }
44
+
45
+ export interface PilotToolResult {
46
+ readonly callId: string
47
+ readonly name: string
48
+ readonly isError: boolean
49
+ readonly content: string
50
+ readonly normalizedError?: string
51
+ readonly resultHash: string
52
+ }
53
+
54
+ export interface PilotTurnSummary {
55
+ readonly iteration: number
56
+ readonly calls: readonly PilotToolCall[]
57
+ readonly results: readonly PilotToolResult[]
58
+ readonly readOnly: boolean
59
+ readonly successfulVirtualWrite: boolean
60
+ readonly milestone: number
61
+ }
62
+
63
+ /** Complete supervisor-owned observation. It contains no oracle labels. */
64
+ export interface PilotBoundary {
65
+ readonly schemaVersion: 1
66
+ readonly experimentId: string
67
+ readonly runId: string
68
+ readonly attemptId: string
69
+ readonly epoch: number
70
+ readonly sequence: number
71
+ readonly iteration: number
72
+ readonly maxIterations: number
73
+ readonly calls: readonly PilotToolCall[]
74
+ readonly results: readonly PilotToolResult[]
75
+ readonly recentTurns: readonly PilotTurnSummary[]
76
+ readonly virtualRevisionHashes: Readonly<Record<string, string>>
77
+ readonly inputTokens: number
78
+ readonly outputTokens: number
79
+ readonly reasoningAvailable: boolean
80
+ readonly truncationFlags: readonly string[]
81
+ readonly deterministicGuardrails: Readonly<{
82
+ consecutiveMistakes: number
83
+ identicalCallStreak: number
84
+ readOnlyStreak: number
85
+ sameToolErrorStreak: number
86
+ repeatedToolFailureStreak: number
87
+ }>
88
+ }
89
+
90
+ export interface Standardization {
91
+ readonly mean: Record<MonitorFeatureName, number>
92
+ readonly scale: Record<MonitorFeatureName, number>
93
+ }
94
+
95
+ export interface Calibration {
96
+ readonly intercept: number
97
+ readonly slope: number
98
+ }
99
+
100
+ export interface FrozenPredictorArtifact {
101
+ readonly schemaVersion: 1
102
+ readonly featureOrder: readonly MonitorFeatureName[]
103
+ readonly coefficients: Record<MonitorFeatureName, number>
104
+ readonly intercept: number
105
+ readonly standardization: Standardization
106
+ readonly calibration: Calibration
107
+ readonly threshold: number
108
+ readonly l2: 1
109
+ readonly modelDigest: string
110
+ readonly templatesHash: string
111
+ readonly artifactHash: string
112
+ }
113
+
114
+ export interface Prediction {
115
+ readonly probability: number
116
+ readonly threshold: number
117
+ readonly features: MonitorFeatureVector
118
+ readonly featureHash: string
119
+ readonly predictorHash: string
120
+ readonly latencyNs: number
121
+ readonly abstentionReason?: "missing" | "non-finite" | "unknown-model" | "malformed"
122
+ }
123
+
124
+ export interface InterventionDecision {
125
+ readonly interventionId: string
126
+ readonly templateId: IterationInterventionTemplateId
127
+ readonly reason: "threshold" | "random-schedule"
128
+ }
129
+
130
+ export interface PilotInterventionRecord extends InterventionDecision {
131
+ readonly boundarySequence: number
132
+ readonly delivered: boolean
133
+ }
134
+
135
+ export interface PilotBudget {
136
+ readonly maxCalls: 16
137
+ readonly maxTokens: 20_000
138
+ readonly maxWallMs: number
139
+ readonly globalMaxTokens: number
140
+ readonly globalMaxWallMs: number
141
+ }
142
+
143
+ export interface PilotManifest {
144
+ readonly schemaVersion: 1
145
+ readonly experimentId: string
146
+ readonly createdAt: string
147
+ readonly modelDigest: string
148
+ readonly modelSettings: Readonly<Record<string, unknown>>
149
+ readonly fixtureHash: string
150
+ readonly splitHash: string
151
+ readonly templatesHash: string
152
+ readonly featureOrder: readonly MonitorFeatureName[]
153
+ readonly predictorHash?: string
154
+ readonly threshold?: number
155
+ readonly armAssignments: readonly PilotArm[]
156
+ readonly seeds: readonly number[]
157
+ readonly budget: PilotBudget
158
+ readonly randomProbability?: number
159
+ readonly randomBoundaryDistribution?: readonly Readonly<{ boundarySequence: number; weight: number }>[]
160
+ }
161
+
162
+ export interface PilotOutcome {
163
+ readonly experimentId: string
164
+ readonly runId: string
165
+ readonly attemptId: string
166
+ readonly epoch: number
167
+ readonly fixtureId: string
168
+ readonly split: FixtureSplit
169
+ readonly arm: PilotArm
170
+ readonly valid: boolean
171
+ readonly success: boolean
172
+ readonly finalArtifactHash: string
173
+ readonly oracleHash: string
174
+ readonly verifierHash: string
175
+ readonly milestoneCount: number
176
+ readonly redundantCalls: number
177
+ readonly invalidRetries: number
178
+ readonly forbiddenEffectAttempts: number
179
+ readonly experimentalInterventionAttempted: boolean
180
+ readonly correctionDelivered: boolean
181
+ readonly deterministicInterventionCount: number
182
+ readonly tokens: number
183
+ readonly calls: number
184
+ readonly elapsedMs: number
185
+ readonly stopReason: string
186
+ readonly predictionRecords: readonly Prediction[]
187
+ readonly interventions: readonly PilotInterventionRecord[]
188
+ readonly scheduledIntervention: boolean
189
+ readonly scheduledBoundarySequence?: number
190
+ /** Labels are retained only in the outcome analysis, never passed to prediction. */
191
+ readonly detectionLabels: readonly (0 | 1)[]
192
+ }