headlesscode 1.2.1 → 1.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +245 -459
- package/package.json +2 -1
- package/shared/openshell/headlesscode-openrouter.yaml +21 -0
- package/shared/openshell/headlesscode-policy.yaml +21 -0
- package/src/cli.ts +30 -0
- package/src/cloud/openshell-preflight.ts +16 -0
- package/src/cloud/openshell-provider.ts +582 -0
- package/src/cloud/openshell-session.ts +119 -0
- package/src/cloud/openshell-subsession.ts +63 -0
- package/src/cloud/openshell-worker.ts +124 -0
- package/src/engine/events.ts +3 -0
- package/src/engine/loop.ts +143 -0
- package/src/engine/types.ts +42 -0
- package/src/monitoring/controller.ts +73 -0
- package/src/monitoring/features.ts +103 -0
- package/src/monitoring/index.ts +4 -0
- package/src/monitoring/predictor.ts +246 -0
- package/src/monitoring/types.ts +192 -0
- package/src/orchestrator/cli.ts +24 -0
- package/src/orchestrator/reviewer.ts +11 -0
- package/src/project-store.ts +11 -0
- package/src/qa/qa.ts +12 -0
- package/src/watcher/cli.ts +18 -0
- package/src/watcher/watch.ts +3 -1
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import { createHash } from "node:crypto"
|
|
2
|
+
import type { MonitorFeatureName, MonitorFeatureVector, PilotBoundary, PilotToolCall, PilotToolResult } from "./types.js"
|
|
3
|
+
import { MONITOR_FEATURE_ORDER } from "./types.js"
|
|
4
|
+
|
|
5
|
+
export function stableJson(value: unknown): string {
|
|
6
|
+
if (value === null || typeof value !== "object") return JSON.stringify(value)
|
|
7
|
+
if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`
|
|
8
|
+
const record = value as Record<string, unknown>
|
|
9
|
+
return `{${Object.keys(record)
|
|
10
|
+
.sort()
|
|
11
|
+
.map((key) => `${JSON.stringify(key)}:${stableJson(record[key])}`)
|
|
12
|
+
.join(",")}}`
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export function sha256(value: unknown): string {
|
|
16
|
+
return createHash("sha256").update(typeof value === "string" ? value : stableJson(value)).digest("hex")
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
function isReadOnlyCall(call: PilotToolCall): boolean {
|
|
20
|
+
return call.name === "read_file" || call.name === "list_files" || call.name === "execute_command"
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function isWriteCall(call: PilotToolCall): boolean {
|
|
24
|
+
return call.name === "apply_diff"
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function callSignature(call: PilotToolCall): string {
|
|
28
|
+
return `${call.name}:${stableJson(call.args)}`
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function resultSignature(result: PilotToolResult): string {
|
|
32
|
+
return `${result.name}:${result.isError}:${result.resultHash}`
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function boundedFraction(numerator: number, denominator: number): number {
|
|
36
|
+
return denominator > 0 ? numerator / denominator : 0
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Extract only current and recent action/result state. Oracle fields never enter here. */
|
|
40
|
+
export function extractMonitorFeatures(boundary: PilotBoundary): MonitorFeatureVector {
|
|
41
|
+
const turns = [...boundary.recentTurns].slice(-4)
|
|
42
|
+
const calls = turns.flatMap((turn) => turn.calls)
|
|
43
|
+
const results = turns.flatMap((turn) => turn.results)
|
|
44
|
+
const lastCall = boundary.calls.at(-1)
|
|
45
|
+
let identicalCallStreak = boundary.deterministicGuardrails.identicalCallStreak
|
|
46
|
+
if (lastCall !== undefined) {
|
|
47
|
+
identicalCallStreak = 0
|
|
48
|
+
for (const call of [...calls].reverse()) {
|
|
49
|
+
if (callSignature(call) !== callSignature(lastCall)) break
|
|
50
|
+
identicalCallStreak++
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
let sameToolErrorStreak = boundary.deterministicGuardrails.sameToolErrorStreak
|
|
54
|
+
const lastError = [...results].reverse().find((result) => result.isError)
|
|
55
|
+
if (lastError !== undefined) {
|
|
56
|
+
sameToolErrorStreak = 0
|
|
57
|
+
for (const result of [...results].reverse()) {
|
|
58
|
+
if (!result.isError || result.name !== lastError.name) break
|
|
59
|
+
sameToolErrorStreak++
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
const distinctCalls = new Set(calls.map(callSignature)).size
|
|
63
|
+
const errorCount = results.filter((result) => result.isError).length
|
|
64
|
+
const repeatedResults = results.length - new Set(results.map(resultSignature)).size
|
|
65
|
+
let turnsSinceSuccessfulVirtualWrite = 0
|
|
66
|
+
for (const turn of [...turns].reverse()) {
|
|
67
|
+
if (turn.successfulVirtualWrite) break
|
|
68
|
+
turnsSinceSuccessfulVirtualWrite++
|
|
69
|
+
}
|
|
70
|
+
const values: Record<MonitorFeatureName, number> = {
|
|
71
|
+
identicalCallStreak,
|
|
72
|
+
sameToolErrorStreak,
|
|
73
|
+
readOnlyStreak: boundary.deterministicGuardrails.readOnlyStreak,
|
|
74
|
+
uniqueCallFraction4: boundedFraction(distinctCalls, calls.length),
|
|
75
|
+
errorFraction4: boundedFraction(errorCount, results.length),
|
|
76
|
+
turnsSinceSuccessfulVirtualWrite,
|
|
77
|
+
repeatedResultFraction4: boundedFraction(repeatedResults, results.length),
|
|
78
|
+
iterationBudgetFraction: boundary.maxIterations > 0 ? boundary.iteration / boundary.maxIterations : 1,
|
|
79
|
+
}
|
|
80
|
+
return values
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export function featureVector(features: MonitorFeatureVector): number[] {
|
|
84
|
+
return MONITOR_FEATURE_ORDER.map((name) => features[name])
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export function featureHash(features: MonitorFeatureVector): string {
|
|
88
|
+
return sha256(featureVector(features))
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
export function assertFiniteFeatures(features: MonitorFeatureVector): void {
|
|
92
|
+
for (const name of MONITOR_FEATURE_ORDER) {
|
|
93
|
+
if (!Number.isFinite(features[name])) throw new Error(`non-finite feature: ${name}`)
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function currentTurnIsReadOnly(boundary: PilotBoundary): boolean {
|
|
98
|
+
return boundary.calls.length > 0 && boundary.calls.every(isReadOnlyCall)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export function currentTurnHasWrite(boundary: PilotBoundary): boolean {
|
|
102
|
+
return boundary.calls.some(isWriteCall) && boundary.results.some((result) => result.name === "apply_diff" && !result.isError)
|
|
103
|
+
}
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
Calibration,
|
|
3
|
+
FrozenPredictorArtifact,
|
|
4
|
+
MonitorFeatureName,
|
|
5
|
+
MonitorFeatureVector,
|
|
6
|
+
PilotBoundary,
|
|
7
|
+
Prediction,
|
|
8
|
+
Standardization,
|
|
9
|
+
} from "./types.js"
|
|
10
|
+
import { MONITOR_FEATURE_ORDER } from "./types.js"
|
|
11
|
+
import { assertFiniteFeatures, extractMonitorFeatures, featureHash, sha256 } from "./features.js"
|
|
12
|
+
|
|
13
|
+
export interface LabeledWindow {
|
|
14
|
+
readonly features: MonitorFeatureVector
|
|
15
|
+
readonly label: 0 | 1
|
|
16
|
+
readonly trajectoryId: string
|
|
17
|
+
readonly boundarySequence?: number
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export interface CalibrationPoint {
|
|
21
|
+
readonly score: number
|
|
22
|
+
readonly label: 0 | 1
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export interface FitOptions {
|
|
26
|
+
readonly modelDigest: string
|
|
27
|
+
readonly templatesHash: string
|
|
28
|
+
readonly l2?: 1
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function sigmoid(value: number): number {
|
|
32
|
+
if (value >= 0) {
|
|
33
|
+
const z = Math.exp(-value)
|
|
34
|
+
return 1 / (1 + z)
|
|
35
|
+
}
|
|
36
|
+
const z = Math.exp(value)
|
|
37
|
+
return z / (1 + z)
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function emptyFeatureRecord(value: number): Record<MonitorFeatureName, number> {
|
|
41
|
+
return Object.fromEntries(MONITOR_FEATURE_ORDER.map((name) => [name, value])) as Record<MonitorFeatureName, number>
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function standardize(rows: readonly LabeledWindow[]): Standardization {
|
|
45
|
+
const mean = emptyFeatureRecord(0)
|
|
46
|
+
const scale = emptyFeatureRecord(1)
|
|
47
|
+
for (const row of rows) {
|
|
48
|
+
for (const name of MONITOR_FEATURE_ORDER) mean[name] += row.features[name]
|
|
49
|
+
}
|
|
50
|
+
for (const name of MONITOR_FEATURE_ORDER) mean[name] /= Math.max(1, rows.length)
|
|
51
|
+
for (const row of rows) {
|
|
52
|
+
for (const name of MONITOR_FEATURE_ORDER) {
|
|
53
|
+
const delta = row.features[name] - mean[name]
|
|
54
|
+
scale[name] += delta * delta
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
for (const name of MONITOR_FEATURE_ORDER) scale[name] = Math.sqrt((scale[name] - 1) / Math.max(1, rows.length)) || 1
|
|
58
|
+
return { mean, scale }
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function standardizedVector(features: MonitorFeatureVector, normalization: Standardization): number[] {
|
|
62
|
+
assertFiniteFeatures(features)
|
|
63
|
+
return MONITOR_FEATURE_ORDER.map((name) => (features[name] - normalization.mean[name]) / normalization.scale[name])
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function fitLogistic(rows: readonly LabeledWindow[], normalization: Standardization): {
|
|
67
|
+
coefficients: Record<MonitorFeatureName, number>
|
|
68
|
+
intercept: number
|
|
69
|
+
} {
|
|
70
|
+
const coefficients = emptyFeatureRecord(0)
|
|
71
|
+
let intercept = 0
|
|
72
|
+
// Fixed optimizer settings are part of the frozen baseline, not tunable data.
|
|
73
|
+
const learningRate = 0.08
|
|
74
|
+
for (let step = 0; step < 2_000; step++) {
|
|
75
|
+
const gradient = emptyFeatureRecord(0)
|
|
76
|
+
let interceptGradient = 0
|
|
77
|
+
for (const row of rows) {
|
|
78
|
+
const x = standardizedVector(row.features, normalization)
|
|
79
|
+
let logit = intercept
|
|
80
|
+
for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) logit += coefficients[MONITOR_FEATURE_ORDER[i]] * x[i]
|
|
81
|
+
const error = sigmoid(logit) - row.label
|
|
82
|
+
interceptGradient += error
|
|
83
|
+
for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) gradient[MONITOR_FEATURE_ORDER[i]] += error * x[i]
|
|
84
|
+
}
|
|
85
|
+
const denominator = Math.max(1, rows.length)
|
|
86
|
+
intercept -= learningRate * (interceptGradient / denominator)
|
|
87
|
+
for (const name of MONITOR_FEATURE_ORDER) {
|
|
88
|
+
// Fixed L2 coefficient 1.0; intercept is not regularized.
|
|
89
|
+
coefficients[name] -= learningRate * (gradient[name] / denominator + coefficients[name] / denominator)
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
return { coefficients, intercept }
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
function rawScore(artifact: Omit<FrozenPredictorArtifact, "artifactHash">, features: MonitorFeatureVector): number {
|
|
96
|
+
const x = standardizedVector(features, artifact.standardization)
|
|
97
|
+
let logit = artifact.intercept
|
|
98
|
+
for (let i = 0; i < MONITOR_FEATURE_ORDER.length; i++) logit += artifact.coefficients[MONITOR_FEATURE_ORDER[i]] * x[i]
|
|
99
|
+
return sigmoid(logit)
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function fitCalibration(points: readonly CalibrationPoint[]): Calibration {
|
|
103
|
+
let intercept = 0
|
|
104
|
+
let slope = 1
|
|
105
|
+
for (let step = 0; step < 1_000; step++) {
|
|
106
|
+
let gi = 0
|
|
107
|
+
let gs = 0
|
|
108
|
+
for (const point of points) {
|
|
109
|
+
const logit = Math.log(Math.min(1 - 1e-6, Math.max(1e-6, point.score)) / Math.max(1e-6, 1 - point.score))
|
|
110
|
+
const probability = sigmoid(intercept + slope * logit)
|
|
111
|
+
const error = probability - point.label
|
|
112
|
+
gi += error
|
|
113
|
+
gs += error * logit
|
|
114
|
+
}
|
|
115
|
+
const denominator = Math.max(1, points.length)
|
|
116
|
+
intercept -= 0.05 * gi / denominator
|
|
117
|
+
slope -= 0.05 * gs / denominator
|
|
118
|
+
}
|
|
119
|
+
return { intercept, slope }
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
function calibratedScore(artifact: Omit<FrozenPredictorArtifact, "artifactHash">, features: MonitorFeatureVector): number {
|
|
123
|
+
const score = rawScore(artifact, features)
|
|
124
|
+
const logit = Math.log(Math.min(1 - 1e-6, Math.max(1e-6, score)) / Math.max(1e-6, 1 - score))
|
|
125
|
+
return sigmoid(artifact.calibration.intercept + artifact.calibration.slope * logit)
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
export function chooseThresholdFromScores(rows: readonly CalibrationPoint[]): number | undefined {
|
|
129
|
+
for (let step = 10; step <= 19; step++) {
|
|
130
|
+
const threshold = step / 20
|
|
131
|
+
const selected = rows.filter((row) => row.score >= threshold)
|
|
132
|
+
const truePositive = selected.filter((row) => row.label === 1).length
|
|
133
|
+
const falsePositive = selected.filter((row) => row.label === 0).length
|
|
134
|
+
const positives = rows.filter((row) => row.label === 1).length
|
|
135
|
+
const negatives = rows.filter((row) => row.label === 0).length
|
|
136
|
+
const precision = selected.length === 0 ? 1 : truePositive / selected.length
|
|
137
|
+
const falsePositiveRate = negatives === 0 ? 0 : falsePositive / negatives
|
|
138
|
+
if (precision >= 0.8 && falsePositiveRate <= 0.1 && positives >= 10) return threshold
|
|
139
|
+
}
|
|
140
|
+
return undefined
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
export function fitFrozenPredictor(
|
|
144
|
+
rows: readonly LabeledWindow[],
|
|
145
|
+
calibrationRows: readonly LabeledWindow[],
|
|
146
|
+
options: FitOptions,
|
|
147
|
+
): FrozenPredictorArtifact {
|
|
148
|
+
if (rows.length === 0) throw new Error("cannot fit predictor without training rows")
|
|
149
|
+
const standardization = standardize(rows)
|
|
150
|
+
const model = fitLogistic(rows, standardization)
|
|
151
|
+
const provisional: Omit<FrozenPredictorArtifact, "artifactHash"> = {
|
|
152
|
+
schemaVersion: 1,
|
|
153
|
+
featureOrder: MONITOR_FEATURE_ORDER,
|
|
154
|
+
coefficients: model.coefficients,
|
|
155
|
+
intercept: model.intercept,
|
|
156
|
+
standardization,
|
|
157
|
+
calibration: fitCalibration(
|
|
158
|
+
calibrationRows.map((row) => ({
|
|
159
|
+
score: rawScore({
|
|
160
|
+
schemaVersion: 1,
|
|
161
|
+
featureOrder: MONITOR_FEATURE_ORDER,
|
|
162
|
+
coefficients: model.coefficients,
|
|
163
|
+
intercept: model.intercept,
|
|
164
|
+
standardization,
|
|
165
|
+
calibration: { intercept: 0, slope: 1 },
|
|
166
|
+
threshold: 0.5,
|
|
167
|
+
l2: 1,
|
|
168
|
+
modelDigest: options.modelDigest,
|
|
169
|
+
templatesHash: options.templatesHash,
|
|
170
|
+
}, row.features),
|
|
171
|
+
label: row.label,
|
|
172
|
+
})),
|
|
173
|
+
),
|
|
174
|
+
threshold: 0.5,
|
|
175
|
+
l2: 1,
|
|
176
|
+
modelDigest: options.modelDigest,
|
|
177
|
+
templatesHash: options.templatesHash,
|
|
178
|
+
}
|
|
179
|
+
const calibrationScores = calibrationRows.map((row) => ({
|
|
180
|
+
score: calibratedScore(provisional, row.features),
|
|
181
|
+
label: row.label,
|
|
182
|
+
}))
|
|
183
|
+
const threshold = chooseThresholdFromScores(calibrationScores)
|
|
184
|
+
if (threshold === undefined) throw new Error("STOP: no qualifying calibration threshold")
|
|
185
|
+
const frozen = { ...provisional, threshold }
|
|
186
|
+
return { ...frozen, artifactHash: sha256(frozen) }
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
export function predict(artifact: FrozenPredictorArtifact, boundary: PilotBoundary): Prediction {
|
|
190
|
+
const started = process.hrtime.bigint()
|
|
191
|
+
try {
|
|
192
|
+
return predictFromFeatures(artifact, extractMonitorFeatures(boundary))
|
|
193
|
+
} catch (error) {
|
|
194
|
+
const elapsedNs = Number(process.hrtime.bigint() - started)
|
|
195
|
+
const features = emptyFeatureRecord(Number.NaN)
|
|
196
|
+
return {
|
|
197
|
+
probability: 0,
|
|
198
|
+
threshold: artifact.threshold,
|
|
199
|
+
features,
|
|
200
|
+
featureHash: sha256(features),
|
|
201
|
+
predictorHash: artifact.artifactHash,
|
|
202
|
+
latencyNs: elapsedNs,
|
|
203
|
+
abstentionReason: error instanceof Error && error.message.includes("finite")
|
|
204
|
+
? "non-finite"
|
|
205
|
+
: error instanceof Error && error.message.includes("unknown model")
|
|
206
|
+
? "unknown-model"
|
|
207
|
+
: "malformed",
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
export function predictFromFeatures(artifact: FrozenPredictorArtifact, features: MonitorFeatureVector): Prediction {
|
|
213
|
+
const started = process.hrtime.bigint()
|
|
214
|
+
try {
|
|
215
|
+
assertFiniteFeatures(features)
|
|
216
|
+
if (artifact.modelDigest.trim() === "") throw new Error("unknown model")
|
|
217
|
+
const provisional = artifact as Omit<FrozenPredictorArtifact, "artifactHash">
|
|
218
|
+
const probability = calibratedScore(provisional, features)
|
|
219
|
+
return {
|
|
220
|
+
probability,
|
|
221
|
+
threshold: artifact.threshold,
|
|
222
|
+
features,
|
|
223
|
+
featureHash: featureHash(features),
|
|
224
|
+
predictorHash: artifact.artifactHash,
|
|
225
|
+
latencyNs: Number(process.hrtime.bigint() - started),
|
|
226
|
+
}
|
|
227
|
+
} catch (error) {
|
|
228
|
+
return {
|
|
229
|
+
probability: 0,
|
|
230
|
+
threshold: artifact.threshold,
|
|
231
|
+
features,
|
|
232
|
+
featureHash: sha256(features),
|
|
233
|
+
predictorHash: artifact.artifactHash,
|
|
234
|
+
latencyNs: Number(process.hrtime.bigint() - started),
|
|
235
|
+
abstentionReason: error instanceof Error && error.message.includes("finite")
|
|
236
|
+
? "non-finite"
|
|
237
|
+
: error instanceof Error && error.message.includes("unknown model")
|
|
238
|
+
? "unknown-model"
|
|
239
|
+
: "malformed",
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
export function predictedIntervention(artifact: FrozenPredictorArtifact, prediction: Prediction): boolean {
|
|
245
|
+
return prediction.abstentionReason === undefined && prediction.probability >= artifact.threshold
|
|
246
|
+
}
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
import type { IterationInterventionTemplateId } from "../engine/types.js"
|
|
2
|
+
|
|
3
|
+
export { type IterationInterventionTemplateId }
|
|
4
|
+
|
|
5
|
+
export const MONITOR_FEATURE_ORDER = [
|
|
6
|
+
"identicalCallStreak",
|
|
7
|
+
"sameToolErrorStreak",
|
|
8
|
+
"readOnlyStreak",
|
|
9
|
+
"uniqueCallFraction4",
|
|
10
|
+
"errorFraction4",
|
|
11
|
+
"turnsSinceSuccessfulVirtualWrite",
|
|
12
|
+
"repeatedResultFraction4",
|
|
13
|
+
"iterationBudgetFraction",
|
|
14
|
+
] as const
|
|
15
|
+
|
|
16
|
+
export type MonitorFeatureName = (typeof MONITOR_FEATURE_ORDER)[number]
|
|
17
|
+
export type MonitorFeatureVector = Record<MonitorFeatureName, number>
|
|
18
|
+
|
|
19
|
+
export type PilotArm = "B" | "M" | "R" | "G"
|
|
20
|
+
export type FixtureFamily = "wrong-key-path" | "revision-mismatch" | "failed-validation" | "legitimate-investigation"
|
|
21
|
+
export type FixtureSplit = "train" | "calibration" | "test"
|
|
22
|
+
|
|
23
|
+
export interface PilotFileSet {
|
|
24
|
+
readonly [relativePath: string]: string
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export interface PilotFixture {
|
|
28
|
+
readonly id: string
|
|
29
|
+
readonly family: FixtureFamily
|
|
30
|
+
readonly split: FixtureSplit
|
|
31
|
+
readonly task: string
|
|
32
|
+
readonly initialFiles: PilotFileSet
|
|
33
|
+
readonly targetFiles: PilotFileSet
|
|
34
|
+
readonly unrelatedPaths: readonly string[]
|
|
35
|
+
readonly validator: (files: PilotFileSet) => { ok: boolean; message: string; milestone: number }
|
|
36
|
+
readonly oracle: (files: PilotFileSet) => { success: boolean; reason: string; milestone: number }
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export interface PilotToolCall {
|
|
40
|
+
readonly id: string
|
|
41
|
+
readonly name: string
|
|
42
|
+
readonly args: Readonly<Record<string, unknown>>
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface PilotToolResult {
|
|
46
|
+
readonly callId: string
|
|
47
|
+
readonly name: string
|
|
48
|
+
readonly isError: boolean
|
|
49
|
+
readonly content: string
|
|
50
|
+
readonly normalizedError?: string
|
|
51
|
+
readonly resultHash: string
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface PilotTurnSummary {
|
|
55
|
+
readonly iteration: number
|
|
56
|
+
readonly calls: readonly PilotToolCall[]
|
|
57
|
+
readonly results: readonly PilotToolResult[]
|
|
58
|
+
readonly readOnly: boolean
|
|
59
|
+
readonly successfulVirtualWrite: boolean
|
|
60
|
+
readonly milestone: number
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** Complete supervisor-owned observation. It contains no oracle labels. */
|
|
64
|
+
export interface PilotBoundary {
|
|
65
|
+
readonly schemaVersion: 1
|
|
66
|
+
readonly experimentId: string
|
|
67
|
+
readonly runId: string
|
|
68
|
+
readonly attemptId: string
|
|
69
|
+
readonly epoch: number
|
|
70
|
+
readonly sequence: number
|
|
71
|
+
readonly iteration: number
|
|
72
|
+
readonly maxIterations: number
|
|
73
|
+
readonly calls: readonly PilotToolCall[]
|
|
74
|
+
readonly results: readonly PilotToolResult[]
|
|
75
|
+
readonly recentTurns: readonly PilotTurnSummary[]
|
|
76
|
+
readonly virtualRevisionHashes: Readonly<Record<string, string>>
|
|
77
|
+
readonly inputTokens: number
|
|
78
|
+
readonly outputTokens: number
|
|
79
|
+
readonly reasoningAvailable: boolean
|
|
80
|
+
readonly truncationFlags: readonly string[]
|
|
81
|
+
readonly deterministicGuardrails: Readonly<{
|
|
82
|
+
consecutiveMistakes: number
|
|
83
|
+
identicalCallStreak: number
|
|
84
|
+
readOnlyStreak: number
|
|
85
|
+
sameToolErrorStreak: number
|
|
86
|
+
repeatedToolFailureStreak: number
|
|
87
|
+
}>
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export interface Standardization {
|
|
91
|
+
readonly mean: Record<MonitorFeatureName, number>
|
|
92
|
+
readonly scale: Record<MonitorFeatureName, number>
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export interface Calibration {
|
|
96
|
+
readonly intercept: number
|
|
97
|
+
readonly slope: number
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
export interface FrozenPredictorArtifact {
|
|
101
|
+
readonly schemaVersion: 1
|
|
102
|
+
readonly featureOrder: readonly MonitorFeatureName[]
|
|
103
|
+
readonly coefficients: Record<MonitorFeatureName, number>
|
|
104
|
+
readonly intercept: number
|
|
105
|
+
readonly standardization: Standardization
|
|
106
|
+
readonly calibration: Calibration
|
|
107
|
+
readonly threshold: number
|
|
108
|
+
readonly l2: 1
|
|
109
|
+
readonly modelDigest: string
|
|
110
|
+
readonly templatesHash: string
|
|
111
|
+
readonly artifactHash: string
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
export interface Prediction {
|
|
115
|
+
readonly probability: number
|
|
116
|
+
readonly threshold: number
|
|
117
|
+
readonly features: MonitorFeatureVector
|
|
118
|
+
readonly featureHash: string
|
|
119
|
+
readonly predictorHash: string
|
|
120
|
+
readonly latencyNs: number
|
|
121
|
+
readonly abstentionReason?: "missing" | "non-finite" | "unknown-model" | "malformed"
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
export interface InterventionDecision {
|
|
125
|
+
readonly interventionId: string
|
|
126
|
+
readonly templateId: IterationInterventionTemplateId
|
|
127
|
+
readonly reason: "threshold" | "random-schedule"
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
export interface PilotInterventionRecord extends InterventionDecision {
|
|
131
|
+
readonly boundarySequence: number
|
|
132
|
+
readonly delivered: boolean
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
export interface PilotBudget {
|
|
136
|
+
readonly maxCalls: 16
|
|
137
|
+
readonly maxTokens: 20_000
|
|
138
|
+
readonly maxWallMs: number
|
|
139
|
+
readonly globalMaxTokens: number
|
|
140
|
+
readonly globalMaxWallMs: number
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
export interface PilotManifest {
|
|
144
|
+
readonly schemaVersion: 1
|
|
145
|
+
readonly experimentId: string
|
|
146
|
+
readonly createdAt: string
|
|
147
|
+
readonly modelDigest: string
|
|
148
|
+
readonly modelSettings: Readonly<Record<string, unknown>>
|
|
149
|
+
readonly fixtureHash: string
|
|
150
|
+
readonly splitHash: string
|
|
151
|
+
readonly templatesHash: string
|
|
152
|
+
readonly featureOrder: readonly MonitorFeatureName[]
|
|
153
|
+
readonly predictorHash?: string
|
|
154
|
+
readonly threshold?: number
|
|
155
|
+
readonly armAssignments: readonly PilotArm[]
|
|
156
|
+
readonly seeds: readonly number[]
|
|
157
|
+
readonly budget: PilotBudget
|
|
158
|
+
readonly randomProbability?: number
|
|
159
|
+
readonly randomBoundaryDistribution?: readonly Readonly<{ boundarySequence: number; weight: number }>[]
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
export interface PilotOutcome {
|
|
163
|
+
readonly experimentId: string
|
|
164
|
+
readonly runId: string
|
|
165
|
+
readonly attemptId: string
|
|
166
|
+
readonly epoch: number
|
|
167
|
+
readonly fixtureId: string
|
|
168
|
+
readonly split: FixtureSplit
|
|
169
|
+
readonly arm: PilotArm
|
|
170
|
+
readonly valid: boolean
|
|
171
|
+
readonly success: boolean
|
|
172
|
+
readonly finalArtifactHash: string
|
|
173
|
+
readonly oracleHash: string
|
|
174
|
+
readonly verifierHash: string
|
|
175
|
+
readonly milestoneCount: number
|
|
176
|
+
readonly redundantCalls: number
|
|
177
|
+
readonly invalidRetries: number
|
|
178
|
+
readonly forbiddenEffectAttempts: number
|
|
179
|
+
readonly experimentalInterventionAttempted: boolean
|
|
180
|
+
readonly correctionDelivered: boolean
|
|
181
|
+
readonly deterministicInterventionCount: number
|
|
182
|
+
readonly tokens: number
|
|
183
|
+
readonly calls: number
|
|
184
|
+
readonly elapsedMs: number
|
|
185
|
+
readonly stopReason: string
|
|
186
|
+
readonly predictionRecords: readonly Prediction[]
|
|
187
|
+
readonly interventions: readonly PilotInterventionRecord[]
|
|
188
|
+
readonly scheduledIntervention: boolean
|
|
189
|
+
readonly scheduledBoundarySequence?: number
|
|
190
|
+
/** Labels are retained only in the outcome analysis, never passed to prediction. */
|
|
191
|
+
readonly detectionLabels: readonly (0 | 1)[]
|
|
192
|
+
}
|
package/src/orchestrator/cli.ts
CHANGED
|
@@ -68,6 +68,7 @@ import {
|
|
|
68
68
|
TRIVIAL_DRIFT_AHEAD,
|
|
69
69
|
} from "./git-sync.js"
|
|
70
70
|
import { resolveModelForMode } from "../config/mode-models.js"
|
|
71
|
+
import { openshellPreflight } from "../cloud/openshell-preflight.js"
|
|
71
72
|
import { runPreflight, runLocalPreflight, type PreflightResult, type LocalPreflightResult } from "../llm/preflight.js"
|
|
72
73
|
import { resolvePerModeEnv } from "../cli.js"
|
|
73
74
|
|
|
@@ -110,6 +111,7 @@ Options:
|
|
|
110
111
|
--poll-interval-ms <n> Watcher poll interval (default: 5000)
|
|
111
112
|
--memory-dir <path> Phase 3 memory dir for workers (passed as HEADLESSCODE_MEMORY_DIR,
|
|
112
113
|
which run-worker.sh forwards as --memory-dir to each worker CLI)
|
|
114
|
+
--execution-provider <local|openshell> Worker runtime (default: local)
|
|
113
115
|
--max-concurrent-sessions <n> Phase 6 GLOBAL cap on concurrent sessions across
|
|
114
116
|
processes (default: $HEADLESSCODE_MAX_CONCURRENT_SESSIONS or 3).
|
|
115
117
|
When the cap is already reached this run ABORTS with a clear
|
|
@@ -195,6 +197,7 @@ Environment:
|
|
|
195
197
|
interface OrchestrateOptions {
|
|
196
198
|
repo: string
|
|
197
199
|
issues: number[]
|
|
200
|
+
executionProvider: "local" | "openshell"
|
|
198
201
|
issuesJson?: string
|
|
199
202
|
/**
|
|
200
203
|
* Fix 3: file a REAL GitHub issue for every synthetic --issues-json entry
|
|
@@ -301,6 +304,7 @@ export function parseOrchestrateArgs(argv: string[]): { options: OrchestrateOpti
|
|
|
301
304
|
const options: OrchestrateOptions = {
|
|
302
305
|
repo: "",
|
|
303
306
|
issues: [],
|
|
307
|
+
executionProvider: process.env.HEADLESSCODE_EXECUTION_PROVIDER === "openshell" ? "openshell" : "local",
|
|
304
308
|
mode: process.env.ORCHESTRATOR_MODE ?? "code",
|
|
305
309
|
reviewMode: "deepseek-reviewer",
|
|
306
310
|
review: true,
|
|
@@ -341,6 +345,12 @@ export function parseOrchestrateArgs(argv: string[]): { options: OrchestrateOpti
|
|
|
341
345
|
}
|
|
342
346
|
|
|
343
347
|
switch (flag) {
|
|
348
|
+
case "--execution-provider": {
|
|
349
|
+
const v = next()
|
|
350
|
+
if (v !== "local" && v !== "openshell") return { options, error: "--execution-provider must be local or openshell" }
|
|
351
|
+
options.executionProvider = v
|
|
352
|
+
break
|
|
353
|
+
}
|
|
344
354
|
case "--repo": {
|
|
345
355
|
const v = next()
|
|
346
356
|
if (v === undefined) {
|
|
@@ -1614,6 +1624,7 @@ export interface BuildSpawnEnvOptions {
|
|
|
1614
1624
|
repo: string
|
|
1615
1625
|
/** Harness mode for the workers (e.g. "code"). */
|
|
1616
1626
|
mode: string
|
|
1627
|
+
executionProvider?: "local" | "openshell"
|
|
1617
1628
|
/** An explicit --model flag value, if the caller passed one — always wins (resolveModelForMode rule 1). */
|
|
1618
1629
|
explicitModel?: string
|
|
1619
1630
|
memoryDir?: string
|
|
@@ -1649,6 +1660,7 @@ export function buildSpawnEnv(options: BuildSpawnEnvOptions): { env: NodeJS.Proc
|
|
|
1649
1660
|
...(options.env ?? process.env),
|
|
1650
1661
|
TARGET_REPO: options.repo,
|
|
1651
1662
|
ORCHESTRATOR_MODE: options.mode,
|
|
1663
|
+
HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider ?? process.env.HEADLESSCODE_EXECUTION_PROVIDER ?? "local",
|
|
1652
1664
|
// The resolved worker model must reach the spawner (and the workers it
|
|
1653
1665
|
// launches) — same pattern as HEADLESSCODE_PROJECT below.
|
|
1654
1666
|
...(workerModel ? { OPENROUTER_MODEL: workerModel } : {}),
|
|
@@ -2545,6 +2557,14 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
|
|
|
2545
2557
|
process.stderr.write(`headlesscode orchestrate: not a git repo: ${repo}\n`)
|
|
2546
2558
|
return 2
|
|
2547
2559
|
}
|
|
2560
|
+
if (!options.dryRun && options.executionProvider === "openshell") {
|
|
2561
|
+
const issue = openshellPreflight()
|
|
2562
|
+
if (issue) {
|
|
2563
|
+
process.stderr.write(`headlesscode orchestrate: OpenShell preflight failed: ${issue}\n`)
|
|
2564
|
+
return 2
|
|
2565
|
+
}
|
|
2566
|
+
}
|
|
2567
|
+
process.env.HEADLESSCODE_EXECUTION_PROVIDER = options.executionProvider
|
|
2548
2568
|
|
|
2549
2569
|
let issues: SplitIssue[]
|
|
2550
2570
|
try {
|
|
@@ -2852,6 +2872,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
|
|
|
2852
2872
|
const { env: spawnEnv, workerModel } = buildSpawnEnv({
|
|
2853
2873
|
repo,
|
|
2854
2874
|
mode: options.mode,
|
|
2875
|
+
executionProvider: options.executionProvider,
|
|
2855
2876
|
explicitModel: options.model,
|
|
2856
2877
|
memoryDir: options.memoryDir,
|
|
2857
2878
|
maxIterations: options.maxIterations,
|
|
@@ -3032,6 +3053,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
|
|
|
3032
3053
|
cwd: repo,
|
|
3033
3054
|
env: {
|
|
3034
3055
|
...process.env,
|
|
3056
|
+
HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider,
|
|
3035
3057
|
HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
|
|
3036
3058
|
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
3037
3059
|
},
|
|
@@ -3221,6 +3243,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
|
|
|
3221
3243
|
cwd: repo,
|
|
3222
3244
|
env: {
|
|
3223
3245
|
...process.env,
|
|
3246
|
+
HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider,
|
|
3224
3247
|
HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
|
|
3225
3248
|
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
3226
3249
|
},
|
|
@@ -3376,6 +3399,7 @@ export async function orchestrateMain(argv: string[]): Promise<number> {
|
|
|
3376
3399
|
cwd: repo,
|
|
3377
3400
|
env: {
|
|
3378
3401
|
...process.env,
|
|
3402
|
+
HEADLESSCODE_EXECUTION_PROVIDER: options.executionProvider,
|
|
3379
3403
|
HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
|
|
3380
3404
|
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
3381
3405
|
},
|
|
@@ -36,6 +36,7 @@ import { getNativeTools } from "../vendor/zoo-code/src/core/prompts/tools/native
|
|
|
36
36
|
import { addCustomInstructions } from "../vendor/zoo-code/src/core/prompts/sections/custom-instructions.js"
|
|
37
37
|
import type { ChatTool, LlmClient, SessionResult } from "../engine/types.js"
|
|
38
38
|
import type { SessionBudget } from "../budget/budget.js"
|
|
39
|
+
import { runOpenShellSubsession } from "../cloud/openshell-subsession.js"
|
|
39
40
|
|
|
40
41
|
/** Default location of the reviewer checklist, relative to the harness repo. */
|
|
41
42
|
export const DEFAULT_REVIEW_PROMPT_PATH = "shared/prompts/review-mode-prompt.md"
|
|
@@ -210,6 +211,16 @@ export async function runReview(options: ReviewOptions): Promise<ReviewResult> {
|
|
|
210
211
|
maxIterations = 200,
|
|
211
212
|
budget,
|
|
212
213
|
} = options
|
|
214
|
+
if (process.env.HEADLESSCODE_EXECUTION_PROVIDER === "openshell" && !llmClient) {
|
|
215
|
+
try {
|
|
216
|
+
return await runOpenShellSubsession<ReviewResult>("review", workspaceRoot, {
|
|
217
|
+
workspaceRoot, mode, model, reviewPromptPath, taskText, issues, maxIterations, budget,
|
|
218
|
+
})
|
|
219
|
+
} catch (error) {
|
|
220
|
+
const message = error instanceof Error ? error.message : String(error)
|
|
221
|
+
return { findings: [`[review session error] ${message}`], verdict: "error", summary: `Review session failed: ${message}` }
|
|
222
|
+
}
|
|
223
|
+
}
|
|
213
224
|
|
|
214
225
|
// 2026-08-27: verified live — a review session couldn't find the `curlee`
|
|
215
226
|
// compiler at all ("curlee runtime is not available in this environment"),
|